From 9ca90ff424b82f74eef0da409c7243d950a9f445 Mon Sep 17 00:00:00 2001 From: saipraneeth <2506664+msaipraneeth@users.noreply.github.com> Date: Fri, 4 Sep 2026 07:26:23 +0100 Subject: [PATCH 1/7] Fix(CMEM-8068): preserve non-ASCII characters in JSON output MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit json.dumps() defaults to ensure_ascii=True, which escaped special characters (e.g. ö, ü) into unicode escape sequences both in entity values built from nested GraphQL response data (create_entity) and in the JSON payload written to the target dataset. --- CHANGELOG.md | 7 +++++++ cmem_plugin_graphql/workflow/graphql.py | 2 +- cmem_plugin_graphql/workflow/utils.py | 2 +- tests/test_graphql.py | 11 ++++++++++- 4 files changed, 19 insertions(+), 3 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index d60123b..6f72326 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -5,6 +5,13 @@ All notable changes to this project will be documented in this file. The format is based on [Keep a Changelog](http://keepachangelog.com/) and this project adheres to [Semantic Versioning](https://semver.org/) +## [Unreleased] + +### Fixed + +- GraphQL query plugin no longer converts non-ASCII characters to unicode escape sequences in + entity values and in the JSON written to the target dataset + ## [6.0.0] 2026-09-02 ### Changed diff --git a/cmem_plugin_graphql/workflow/graphql.py b/cmem_plugin_graphql/workflow/graphql.py index 4210294..1ec941b 100644 --- a/cmem_plugin_graphql/workflow/graphql.py +++ b/cmem_plugin_graphql/workflow/graphql.py @@ -211,7 +211,7 @@ def execute(self, inputs: Sequence[Entities], context: ExecutionContext) -> Enti if dataset_id: write_to_dataset( dataset_id, - io.StringIO(json.dumps(payload, indent=2)), + io.StringIO(json.dumps(payload, indent=2, ensure_ascii=False)), context=context.user, ) diff --git a/cmem_plugin_graphql/workflow/utils.py b/cmem_plugin_graphql/workflow/utils.py index 0898b33..fe52cd5 100644 --- a/cmem_plugin_graphql/workflow/utils.py +++ b/cmem_plugin_graphql/workflow/utils.py @@ -63,6 +63,6 @@ def create_entity(paths: list[str], dict_: dict[str, Any]) -> Entity: elif type(value) in (int, float, bool, str): values.append([value]) else: - values.append([json.dumps(value)]) + values.append([json.dumps(value, ensure_ascii=False)]) entity_uri = f"urn:uuid:{uuid.uuid4()!s}" return Entity(uri=entity_uri, values=values) diff --git a/tests/test_graphql.py b/tests/test_graphql.py index 9592ba0..2f0a3a6 100644 --- a/tests/test_graphql.py +++ b/tests/test_graphql.py @@ -21,7 +21,7 @@ from requests import HTTPError from cmem_plugin_graphql.workflow.graphql import GraphQLPlugin -from cmem_plugin_graphql.workflow.utils import is_jinja_template +from cmem_plugin_graphql.workflow.utils import create_entity, is_jinja_template GRAPHQL_URL = "https://cmem-plugin-graphql-test.netlify.app/graphql" @@ -320,6 +320,15 @@ def test_mutation_with_jinja_template(project: str) -> None: assert graphql_response == str(result[0]) +def test_create_entity_keeps_unicode_characters_in_nested_values() -> None: + """Test that non-ASCII characters in a nested (dict/list) value are not escaped""" + entity = create_entity(["nested"], {"nested": {"city": "Köln", "name": "Müller"}}) + value = entity.values[0][0] + assert "\\u00f6" not in value + assert "\\u00fc" not in value + assert json.loads(value) == {"city": "Köln", "name": "Müller"} + + def test_is_string_jinja_template() -> None: """Test plugin execution""" query = "query allFruits($id:ID!) { fruit(id:$id) { id scientific_name } }" From 878272f7e9eaf7c41ed646063793d78b036fba14 Mon Sep 17 00:00:00 2001 From: saipraneeth <2506664+msaipraneeth@users.noreply.github.com> Date: Fri, 4 Sep 2026 07:51:41 +0100 Subject: [PATCH 2/7] Add end-to-end regression test for unicode-safe dataset output MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The public test endpoint's fruit id 4 (Limón / Rutáceae) genuinely contains non-ASCII characters, so it can drive a real end-to-end test of the write_to_dataset() JSON path instead of only the isolated create_entity() unit test. Checks the raw file bytes (not just the json.loads()'d result, which would hide the bug either way) for the absence of the escape sequences. --- tests/test_graphql.py | 29 +++++++++++++++++++++++++---- 1 file changed, 25 insertions(+), 4 deletions(-) diff --git a/tests/test_graphql.py b/tests/test_graphql.py index 2f0a3a6..e8fe756 100644 --- a/tests/test_graphql.py +++ b/tests/test_graphql.py @@ -110,12 +110,15 @@ def _get_client() -> Client: return Client.from_context(context=TestExecutionContext()) -def _read_resource(project_name: str, filename: str) -> Any: # noqa: ANN401 - """Read a JSON resource from a CMEM project.""" +def _read_raw_resource(project_name: str, filename: str) -> str: + """Read the raw (undecoded) text content of a resource from a CMEM project.""" client = _get_client() - content = client.files.read(f"{project_name}:{filename}") + return client.files.read(f"{project_name}:{filename}").decode("utf-8") - return json.loads(content) + +def _read_resource(project_name: str, filename: str) -> Any: # noqa: ANN401 + """Read a JSON resource from a CMEM project.""" + return json.loads(_read_raw_resource(project_name, filename)) @pytest.fixture(scope="module") @@ -157,6 +160,24 @@ def test_execution(project: str) -> None: assert graphql_response == str(result[0]) +@needs_cmem +def test_execution_preserves_unicode_characters(project: str) -> None: + """Test that non-ASCII characters from the GraphQL response are not escaped in the dataset""" + _ = project + query = "query{fruit(id:4){id,scientific_name,fruit_name,family}}" + + plugin = GraphQLPlugin( + graphql_url=GRAPHQL_URL, graphql_query=query, graphql_dataset=DATASET_NAME + ) + plugin.execute([], TestExecutionContext(project_id=PROJECT_NAME)) + raw_content = _read_raw_resource(PROJECT_NAME, RESOURCE_NAME) + assert "\\u00f3" not in raw_content + assert "\\u00e1" not in raw_content + fruit = json.loads(raw_content)[0]["fruit"] + assert fruit["fruit_name"] == "Limón" + assert fruit["family"] == "Rutáceae" + + @needs_cmem def test_execution_with_variables(project: str) -> None: """Test plugin execution""" From 29b5e5ee38a25b9b4aa04758fa6caccb7589b4b5 Mon Sep 17 00:00:00 2001 From: saipraneeth <2506664+msaipraneeth@users.noreply.github.com> Date: Fri, 4 Sep 2026 09:57:36 +0100 Subject: [PATCH 3/7] Remove dead code: get_entities_from_list / create_entity Neither function is called from GraphQLPlugin.execute() or anywhere else in the package - the plugin builds entities via cmem-plugin-base's build_entities_from_data() instead. Confirmed no external callers. --- cmem_plugin_graphql/workflow/utils.py | 46 +-------------------------- tests/test_graphql.py | 11 +------ 2 files changed, 2 insertions(+), 55 deletions(-) diff --git a/cmem_plugin_graphql/workflow/utils.py b/cmem_plugin_graphql/workflow/utils.py index fe52cd5..b750c81 100644 --- a/cmem_plugin_graphql/workflow/utils.py +++ b/cmem_plugin_graphql/workflow/utils.py @@ -1,17 +1,9 @@ """Utils module""" -import json -import uuid from collections.abc import Iterator -from typing import Any import jinja2 -from cmem_plugin_base.dataintegration.entity import ( - Entities, - Entity, - EntityPath, - EntitySchema, -) +from cmem_plugin_base.dataintegration.entity import Entities def get_dict(entities: Entities) -> Iterator[dict[str, str]]: @@ -30,39 +22,3 @@ def is_jinja_template(value: str) -> bool: template = environment.from_string(value) res = template.render() return res != value - - -def get_entities_from_list(data: list[dict[str, Any]]) -> Entities: - """Generate entities from list""" - paths: list[str] = [] - unique_paths: set[str] = set() - entities = [] - # first pass to extract paths - for dict_ in data: - unique_paths.update(set(dict_.keys())) - - paths = list(unique_paths) - for dict_ in data: - entity = create_entity(paths, dict_) - entities.append(entity) - - schema = EntitySchema( - type_uri="https://example.org/vocab/RandomValueRow", - paths=[EntityPath(path=path) for path in paths], - ) - return Entities(entities=entities, schema=schema) - - -def create_entity(paths: list[str], dict_: dict[str, Any]) -> Entity: - """Create entity from dict based on order from paths list""" - values: list[list[str | None]] = [] - for path in paths: - value = dict_.get(path) - if value is None: - values.append([]) - elif type(value) in (int, float, bool, str): - values.append([value]) - else: - values.append([json.dumps(value, ensure_ascii=False)]) - entity_uri = f"urn:uuid:{uuid.uuid4()!s}" - return Entity(uri=entity_uri, values=values) diff --git a/tests/test_graphql.py b/tests/test_graphql.py index e8fe756..4296f59 100644 --- a/tests/test_graphql.py +++ b/tests/test_graphql.py @@ -21,7 +21,7 @@ from requests import HTTPError from cmem_plugin_graphql.workflow.graphql import GraphQLPlugin -from cmem_plugin_graphql.workflow.utils import create_entity, is_jinja_template +from cmem_plugin_graphql.workflow.utils import is_jinja_template GRAPHQL_URL = "https://cmem-plugin-graphql-test.netlify.app/graphql" @@ -341,15 +341,6 @@ def test_mutation_with_jinja_template(project: str) -> None: assert graphql_response == str(result[0]) -def test_create_entity_keeps_unicode_characters_in_nested_values() -> None: - """Test that non-ASCII characters in a nested (dict/list) value are not escaped""" - entity = create_entity(["nested"], {"nested": {"city": "Köln", "name": "Müller"}}) - value = entity.values[0][0] - assert "\\u00f6" not in value - assert "\\u00fc" not in value - assert json.loads(value) == {"city": "Köln", "name": "Müller"} - - def test_is_string_jinja_template() -> None: """Test plugin execution""" query = "query allFruits($id:ID!) { fruit(id:$id) { id scientific_name } }" From 4905b40831000cfe75467cc101f560fad372b223 Mon Sep 17 00:00:00 2001 From: saipraneeth <2506664+msaipraneeth@users.noreply.github.com> Date: Mon, 7 Sep 2026 06:24:31 +0100 Subject: [PATCH 4/7] Expand GraphQL query plugin documentation The @Plugin documentation previously said nothing about the ports, the Jinja-templating value shapes, or what happens when Query/Query variables contains Jinja syntax but no input is connected (a crash for Query, a silent no-op for Query variables). Also corrected the earlier CHANGELOG Fixed entry, which still described the entity-values path that turned out to be dead code and was removed. --- CHANGELOG.md | 7 +++- cmem_plugin_graphql/workflow/graphql.py | 48 ++++++++++++++++++++----- 2 files changed, 45 insertions(+), 10 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 6f72326..d447ec8 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -7,10 +7,15 @@ The format is based on [Keep a Changelog](http://keepachangelog.com/) and this p ## [Unreleased] +### Changed + +- Expanded task and parameter documentation to describe port/value shapes and the Jinja + templating edge cases around missing input + ### Fixed - GraphQL query plugin no longer converts non-ASCII characters to unicode escape sequences in - entity values and in the JSON written to the target dataset + the JSON written to the target dataset ## [6.0.0] 2026-09-02 diff --git a/cmem_plugin_graphql/workflow/graphql.py b/cmem_plugin_graphql/workflow/graphql.py index 1ec941b..61907bf 100644 --- a/cmem_plugin_graphql/workflow/graphql.py +++ b/cmem_plugin_graphql/workflow/graphql.py @@ -33,14 +33,37 @@ @Plugin( label="GraphQL query", - description="Executes a custom GraphQL query to a GraphQL endpoint" - " and saves result to a JSON dataset.", - documentation="""This workflow task performs GraphQL operations by sending - queries, mutations, and variables over operations. Allows for customization - in the GraphQL query using, Jinja queries and Jinja variables, which can be - obtained from entities. The result of the query is saved as a JSON document - in a pre-created JSON dataset. - """, + description="Sends a GraphQL query or mutation to an endpoint and returns the result," + " or writes it to a JSON dataset.", + documentation="""This task sends a GraphQL query or mutation to an endpoint and +captures the response. + +An input port accepts entities, but it only changes anything when **Query** or +**Query variables** actually contains Jinja syntax: the query and variables are +then rendered once per input entity and the endpoint is called once per entity, +with a failed entity logged and skipped rather than failing the whole task. A +purely static query and variables text runs exactly once and ignores any +connected input entirely. + +When **Target JSON Dataset** is left empty, the collected response(s) become +entities returned on the output port, one per query execution. When it is set, +the output port disappears instead and the same responses are written there as +a single JSON array. + +The task typically starts a chain that begins at a GraphQL API and lands the +result either in a downstream transform, via the output port, or in a JSON +dataset for later use. + +A Jinja-templated **Query** or **Query variables** is never checked for GraphQL +syntax errors until it is actually rendered - a mistake in it only surfaces at +runtime, as a failed entity, rather than as a configuration error when the task +is set up. If **Query** contains Jinja syntax and no input is connected at all, +the unrendered `{{ ... }}` text is sent to the GraphQL library as literal +syntax and the task fails outright. If only **Query variables** contains Jinja +syntax while **Query** is static, and no input is connected, the task instead +sends nothing and completes as if zero entities were processed, without +warning that the variables were never rendered. +""", parameters=[ PluginParameter( name="graphql_url", @@ -61,6 +84,9 @@ GraphQL is a query language for APIs and a runtime for fulfilling those queries with your existing data. Learn more on GraphQL [here](https://graphql.org/). +May also contain Jinja syntax (e.g. `{{ id }}`), which is rendered against each +input entity before the query is sent. + Example Query: query allFruits { fruits { id @@ -81,6 +107,9 @@ label="Query variables", description="""Pass dynamic variables when making a query or mutation. +May also contain Jinja syntax (e.g. `{"id": {{ id }}}`), which is rendered +against each input entity before the query is sent. + Example Variables: {"id" : 1} """, default_value="{}", @@ -89,7 +118,8 @@ PluginParameter( name="graphql_dataset", label="Target JSON Dataset", - description="The Dataset where this task will save the JSON results.", + description="The JSON dataset the result is written to. When set, the output port" + " is removed and the result is only available in the dataset.", param_type=DatasetParameterType(dataset_type="json"), advanced=True, default_value="", From 886ebf370c706d4859ea8ccd41833f1f7259fd0e Mon Sep 17 00:00:00 2001 From: saipraneeth <2506664+msaipraneeth@users.noreply.github.com> Date: Mon, 7 Sep 2026 06:41:29 +0100 Subject: [PATCH 5/7] Fix mypy no-any-return in _read_raw_resource cmem_client ships no py.typed marker, so mypy treats client.files.read() as returning Any regardless of its actual bytes return type. An explicit bytes annotation on the intermediate variable lets mypy trust it, so .decode("utf-8") resolves to str as declared. --- tests/test_graphql.py | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/tests/test_graphql.py b/tests/test_graphql.py index 4296f59..0da7d2f 100644 --- a/tests/test_graphql.py +++ b/tests/test_graphql.py @@ -113,7 +113,8 @@ def _get_client() -> Client: def _read_raw_resource(project_name: str, filename: str) -> str: """Read the raw (undecoded) text content of a resource from a CMEM project.""" client = _get_client() - return client.files.read(f"{project_name}:{filename}").decode("utf-8") + content: bytes = client.files.read(f"{project_name}:{filename}") + return content.decode("utf-8") def _read_resource(project_name: str, filename: str) -> Any: # noqa: ANN401 From 78356528393f1e94cc93781f354889983aaa8469 Mon Sep 17 00:00:00 2001 From: saipraneeth <2506664+msaipraneeth@users.noreply.github.com> Date: Mon, 7 Sep 2026 08:02:50 +0100 Subject: [PATCH 6/7] Fix dataset upload truncation for non-ASCII JSON content write_to_dataset()/post_file_resource() are documented and typed to take a byte stream, but the target-dataset write passed an io.StringIO text stream instead. This masqueraded as working while ensure_ascii=True guaranteed pure-ASCII output (1 char == 1 byte); now that real UTF-8 multi-byte characters (e.g. from the ensure_ascii=False fix) flow through, the byte/char-length mismatch silently truncated the uploaded file, as seen by CI's non-ASCII regression test failing with a JSONDecodeError on a file missing its closing bracket. Switched to io.BytesIO(...encode("utf-8")), matching the pattern already used correctly elsewhere (e.g. cmem-plugin-jq, cmem-plugin-validation). --- CHANGELOG.md | 4 ++++ cmem_plugin_graphql/workflow/graphql.py | 2 +- 2 files changed, 5 insertions(+), 1 deletion(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index d447ec8..a5cfd86 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -16,6 +16,10 @@ The format is based on [Keep a Changelog](http://keepachangelog.com/) and this p - GraphQL query plugin no longer converts non-ASCII characters to unicode escape sequences in the JSON written to the target dataset +- Fixed a related upload bug this uncovered: the target dataset write passed a text stream to + an API that expects a byte stream, which silently truncated the uploaded file whenever the + JSON contained multi-byte UTF-8 characters (masked previously only because escaped output + was always pure ASCII) ## [6.0.0] 2026-09-02 diff --git a/cmem_plugin_graphql/workflow/graphql.py b/cmem_plugin_graphql/workflow/graphql.py index 61907bf..93439f4 100644 --- a/cmem_plugin_graphql/workflow/graphql.py +++ b/cmem_plugin_graphql/workflow/graphql.py @@ -241,7 +241,7 @@ def execute(self, inputs: Sequence[Entities], context: ExecutionContext) -> Enti if dataset_id: write_to_dataset( dataset_id, - io.StringIO(json.dumps(payload, indent=2, ensure_ascii=False)), + io.BytesIO(json.dumps(payload, indent=2, ensure_ascii=False).encode("utf-8")), context=context.user, ) From f552278c0b053c506f34df5e19e034d9b8459d6b Mon Sep 17 00:00:00 2001 From: saipraneeth <2506664+msaipraneeth@users.noreply.github.com> Date: Tue, 8 Sep 2026 12:25:46 +0100 Subject: [PATCH 7/7] Fix indented code-block rendering in Query variables description The example line kept the docstring's source indentation, which Markdown renders as an indented code block instead of flowing text. Claude-Session: https://claude.ai/code/session_01XE4Ptbbj45VPRQnPsR9Lvm --- CHANGELOG.md | 2 ++ cmem_plugin_graphql/workflow/graphql.py | 4 ++-- 2 files changed, 4 insertions(+), 2 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index a5cfd86..0f9652a 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -20,6 +20,8 @@ The format is based on [Keep a Changelog](http://keepachangelog.com/) and this p an API that expects a byte stream, which silently truncated the uploaded file whenever the JSON contained multi-byte UTF-8 characters (masked previously only because escaped output was always pure ASCII) +- Query variables parameter description no longer renders its example as an indented code + block due to leftover indentation in the source string ## [6.0.0] 2026-09-02 diff --git a/cmem_plugin_graphql/workflow/graphql.py b/cmem_plugin_graphql/workflow/graphql.py index 93439f4..dd69469 100644 --- a/cmem_plugin_graphql/workflow/graphql.py +++ b/cmem_plugin_graphql/workflow/graphql.py @@ -110,8 +110,8 @@ May also contain Jinja syntax (e.g. `{"id": {{ id }}}`), which is rendered against each input entity before the query is sent. - Example Variables: {"id" : 1} - """, +Example Variables: `{"id" : 1}` +""", default_value="{}", param_type=MultilineStringParameterType(), ),