Skip to content
GitLab
Projects
Groups
Snippets
/
Help
Help
Support
Community forum
Keyboard shortcuts
?
Submit feedback
Sign in
Toggle navigation
Menu
Open sidebar
math_tutor_dev
public_math_tutor
Commits
5dac10a7
Commit
5dac10a7
authored
May 13, 2026
by
Kantz
Browse files
umstellung von json auf yaml
parent
3a4e6b7b
Changes
4
Hide whitespace changes
Inline
Side-by-side
math-tutor/backend/app/deterministic_services/orchestrators/orchestrator_task.py
View file @
5dac10a7
...
@@ -41,7 +41,7 @@ def _ensure_context_task_fields(state: base.ChatState, query_text: str) -> tuple
...
@@ -41,7 +41,7 @@ def _ensure_context_task_fields(state: base.ChatState, query_text: str) -> tuple
base
.
append_tool_log
(
base
.
append_tool_log
(
state
.
tool_log
,
state
.
tool_log
,
"task_
json
_selected"
,
"task_
yaml
_selected"
,
{
"tasks_dir"
:
str
(
task_catalog
.
TASKS_DIR
)},
{
"tasks_dir"
:
str
(
task_catalog
.
TASKS_DIR
)},
{
{
"file"
:
task_file
.
get
(
"_path"
,
""
),
"file"
:
task_file
.
get
(
"_path"
,
""
),
...
...
math-tutor/backend/app/deterministic_services/task_catalog.py
View file @
5dac10a7
from
__future__
import
annotations
from
__future__
import
annotations
import
json
import
re
import
re
import
unicodedata
import
unicodedata
from
functools
import
lru_cache
from
functools
import
lru_cache
from
pathlib
import
Path
from
pathlib
import
Path
from
urllib.parse
import
quote
from
typing
import
Any
from
typing
import
Any
from
urllib.parse
import
quote
import
yaml
from
app.deterministic_services
import
context_store
from
app.deterministic_services
import
context_store
from
app.deterministic_services.vector_store
import
parse_markdown_with_frontmatter
from
app.deterministic_services.vector_store
import
parse_markdown_with_frontmatter
...
@@ -14,7 +15,7 @@ from app.deterministic_services.vector_store import parse_markdown_with_frontmat
...
@@ -14,7 +15,7 @@ from app.deterministic_services.vector_store import parse_markdown_with_frontmat
TASKS_DIR
=
Path
(
__file__
).
resolve
().
parents
[
2
]
/
"sources"
/
"tasks"
TASKS_DIR
=
Path
(
__file__
).
resolve
().
parents
[
2
]
/
"sources"
/
"tasks"
TASK_IMAGES_DIR
=
TASKS_DIR
/
"images"
TASK_IMAGES_DIR
=
TASKS_DIR
/
"images"
SOURCES_DIR
=
Path
(
__file__
).
resolve
().
parents
[
2
]
/
"sources"
SOURCES_DIR
=
Path
(
__file__
).
resolve
().
parents
[
2
]
/
"sources"
SUBSECTION_MAP_PATH
=
TASKS_DIR
/
"_subsection_map.
json
"
SUBSECTION_MAP_PATH
=
TASKS_DIR
/
"_subsection_map.
yaml
"
TASK_ASSET_URL_PREFIX
=
"/api/tasks/assets"
TASK_ASSET_URL_PREFIX
=
"/api/tasks/assets"
def
_normalize_text
(
value
:
str
)
->
str
:
def
_normalize_text
(
value
:
str
)
->
str
:
...
@@ -34,7 +35,7 @@ def _normalize_subsection_key(value: str) -> str:
...
@@ -34,7 +35,7 @@ def _normalize_subsection_key(value: str) -> str:
def
_parse_subsection_ref
(
value
:
str
)
->
tuple
[
int
,
int
,
int
]
|
None
:
def
_parse_subsection_ref
(
value
:
str
)
->
tuple
[
int
,
int
,
int
]
|
None
:
match
=
re
.
match
(
r
"^\s*(\d+)\s*[
:
.]\s*(\d+)\s*[
:
.]\s*(\d+)\s*$"
,
str
(
value
))
match
=
re
.
match
(
r
"^\s*(\d+)\s*[
-
.]\s*(\d+)\s*[
-
.]\s*(\d+)\s*$"
,
str
(
value
))
if
not
match
:
if
not
match
:
return
None
return
None
return
int
(
match
.
group
(
1
)),
int
(
match
.
group
(
2
)),
int
(
match
.
group
(
3
))
return
int
(
match
.
group
(
1
)),
int
(
match
.
group
(
2
)),
int
(
match
.
group
(
3
))
...
@@ -44,7 +45,7 @@ def load_subsection_map(path: Path = SUBSECTION_MAP_PATH) -> dict[str, tuple[int
...
@@ -44,7 +45,7 @@ def load_subsection_map(path: Path = SUBSECTION_MAP_PATH) -> dict[str, tuple[int
if
not
path
.
exists
():
if
not
path
.
exists
():
return
{}
return
{}
try
:
try
:
content
=
json
.
load
s
(
path
.
read_text
(
encoding
=
"utf-8"
))
content
=
yaml
.
safe_
load
(
path
.
read_text
(
encoding
=
"utf-8"
))
except
Exception
:
except
Exception
:
return
{}
return
{}
if
not
isinstance
(
content
,
dict
):
if
not
isinstance
(
content
,
dict
):
...
@@ -186,23 +187,144 @@ def _match_score(query_text: str, candidate_text: str) -> int:
...
@@ -186,23 +187,144 @@ def _match_score(query_text: str, candidate_text: str) -> int:
return
len
(
query_tokens
.
intersection
(
candidate_tokens
))
return
len
(
query_tokens
.
intersection
(
candidate_tokens
))
def
_slugify
(
value
:
str
)
->
str
:
normalized
=
unicodedata
.
normalize
(
"NFKD"
,
str
(
value
))
ascii_value
=
normalized
.
encode
(
"ascii"
,
"ignore"
).
decode
(
"ascii"
)
lowered
=
ascii_value
.
strip
().
lower
()
collapsed
=
re
.
sub
(
r
"[^a-z0-9]+"
,
"-"
,
lowered
)
collapsed
=
re
.
sub
(
r
"-{2,}"
,
"-"
,
collapsed
)
return
collapsed
.
strip
(
"-"
)
def
_normalize_block_images
(
images_raw
:
list
[
dict
[
str
,
str
]])
->
list
[
dict
[
str
,
str
]]:
normalized_images
:
list
[
dict
[
str
,
str
]]
=
[]
for
item
in
images_raw
:
if
not
isinstance
(
item
,
dict
):
continue
src
=
str
(
item
.
get
(
"src"
,
""
)).
strip
().
replace
(
"
\\
"
,
"/"
)
description
=
str
(
item
.
get
(
"description"
,
""
)
or
item
.
get
(
"alt"
,
""
)).
strip
()
if
not
src
or
not
description
:
continue
if
src
.
startswith
(
"/"
)
or
Path
(
src
).
is_absolute
():
continue
normalized_path
=
Path
(
src
)
if
any
(
part
in
{
""
,
"."
,
".."
}
for
part
in
normalized_path
.
parts
):
continue
encoded_src
=
"/"
.
join
(
quote
(
part
)
for
part
in
normalized_path
.
parts
)
normalized_images
.
append
(
{
"src"
:
f
"
{
TASK_ASSET_URL_PREFIX
}
/
{
encoded_src
}
"
,
"description"
:
description
,
}
)
return
normalized_images
def
_extract_text_and_images
(
blocks
:
Any
)
->
tuple
[
list
[
str
],
list
[
dict
[
str
,
str
]]]:
if
not
isinstance
(
blocks
,
list
):
return
[],
[]
text_parts
:
list
[
str
]
=
[]
images
:
list
[
dict
[
str
,
str
]]
=
[]
for
block
in
blocks
:
if
not
isinstance
(
block
,
dict
):
continue
block_type
=
str
(
block
.
get
(
"type"
,
""
)).
strip
().
lower
()
if
block_type
==
"text"
:
text
=
str
(
block
.
get
(
"text"
,
""
)).
strip
()
if
text
:
text_parts
.
append
(
text
)
continue
if
block_type
==
"image"
:
src
=
str
(
block
.
get
(
"src"
,
""
)).
strip
()
alt
=
str
(
block
.
get
(
"alt"
,
""
)).
strip
()
if
src
and
alt
:
images
.
append
({
"src"
:
src
,
"description"
:
alt
})
return
text_parts
,
images
def
_normalize_yaml_task_entry
(
task_entry
:
Any
,
position
:
int
)
->
dict
[
str
,
Any
]
|
None
:
if
not
isinstance
(
task_entry
,
dict
):
return
None
statement_parts
,
statement_images
=
_extract_text_and_images
(
task_entry
.
get
(
"aufgabe"
,
[]))
hint_parts
,
hint_images
=
_extract_text_and_images
(
task_entry
.
get
(
"hinweise"
,
[]))
solution_parts
,
_
=
_extract_text_and_images
(
task_entry
.
get
(
"loesung"
,
[]))
return
{
"id"
:
str
(
position
).
zfill
(
2
),
"statement"
:
"
\n
"
.
join
(
statement_parts
).
strip
(),
"hints"
:
hint_parts
,
"solution"
:
"
\n
"
.
join
(
solution_parts
).
strip
(),
"images"
:
statement_images
+
hint_images
,
}
def
_normalize_exercise_topics
(
raw_topics
:
Any
)
->
list
[
str
]:
if
not
isinstance
(
raw_topics
,
list
):
return
[]
topics
:
list
[
str
]
=
[]
for
item
in
raw_topics
:
topic_id
=
""
if
isinstance
(
item
,
dict
):
topic_id
=
str
(
item
.
get
(
"topic-id"
,
""
)).
strip
()
elif
isinstance
(
item
,
str
):
topic_id
=
item
.
strip
()
if
topic_id
:
topics
.
append
(
topic_id
)
return
topics
def
_normalize_yaml_task_file
(
content
:
Any
,
path
:
Path
)
->
dict
[
str
,
Any
]
|
None
:
if
not
isinstance
(
content
,
dict
):
return
None
if
not
isinstance
(
content
.
get
(
"tasks"
),
list
):
return
None
title
=
str
(
content
.
get
(
"titel"
,
""
)).
strip
()
intro
=
str
(
content
.
get
(
"description"
,
""
)).
strip
()
file_id
=
str
(
content
.
get
(
"id"
,
""
)).
strip
()
or
path
.
stem
slug
=
str
(
content
.
get
(
"slug"
,
""
)).
strip
()
or
_slugify
(
title
or
file_id
)
level
=
str
(
content
.
get
(
"level"
,
""
)).
strip
()
subsections
=
_normalize_exercise_topics
(
content
.
get
(
"exercise_topic"
,
[]))
if
not
title
or
not
file_id
or
not
slug
or
not
level
:
return
None
tasks
:
list
[
dict
[
str
,
Any
]]
=
[]
for
index
,
raw_task
in
enumerate
(
content
.
get
(
"tasks"
,
[]),
start
=
1
):
normalized_task
=
_normalize_yaml_task_entry
(
raw_task
,
index
)
if
normalized_task
is
None
:
return
None
tasks
.
append
(
normalized_task
)
return
{
"_path"
:
str
(
path
),
"_file_id"
:
file_id
,
"title"
:
title
,
"intro"
:
intro
,
"slug"
:
slug
,
"level"
:
level
,
"subsections"
:
subsections
,
"tasks"
:
tasks
,
}
def
load_task_files
(
tasks_dir
:
Path
=
TASKS_DIR
)
->
list
[
dict
[
str
,
Any
]]:
def
load_task_files
(
tasks_dir
:
Path
=
TASKS_DIR
)
->
list
[
dict
[
str
,
Any
]]:
if
not
tasks_dir
.
exists
():
if
not
tasks_dir
.
exists
():
return
[]
return
[]
loaded
:
list
[
dict
[
str
,
Any
]]
=
[]
loaded
:
list
[
dict
[
str
,
Any
]]
=
[]
for
path
in
sorted
(
tasks_dir
.
glob
(
"*.
json
"
)):
for
path
in
sorted
(
[
*
tasks_dir
.
glob
(
"*.
yaml"
),
*
tasks_dir
.
glob
(
"*.yml
"
)
]
):
try
:
try
:
content
=
json
.
load
s
(
path
.
read_text
(
encoding
=
"utf-8"
))
content
=
yaml
.
safe_
load
(
path
.
read_text
(
encoding
=
"utf-8"
))
except
Exception
:
except
Exception
:
continue
continue
if
not
isinstance
(
content
,
dict
):
normalized
=
_normalize_yaml_task_file
(
content
,
path
)
if
normalized
is
None
:
continue
continue
if
not
isinstance
(
content
.
get
(
"tasks"
),
list
):
loaded
.
append
(
normalized
)
continue
content
[
"_path"
]
=
str
(
path
)
content
[
"_file_id"
]
=
path
.
stem
loaded
.
append
(
content
)
return
loaded
return
loaded
...
@@ -226,28 +348,7 @@ def _normalize_task_images(task_entry: dict[str, Any]) -> list[dict[str, str]]:
...
@@ -226,28 +348,7 @@ def _normalize_task_images(task_entry: dict[str, Any]) -> list[dict[str, str]]:
images_raw
=
task_entry
.
get
(
"images"
,
[])
images_raw
=
task_entry
.
get
(
"images"
,
[])
if
not
isinstance
(
images_raw
,
list
):
if
not
isinstance
(
images_raw
,
list
):
return
[]
return
[]
return
_normalize_block_images
(
images_raw
)
normalized_images
:
list
[
dict
[
str
,
str
]]
=
[]
for
item
in
images_raw
:
if
not
isinstance
(
item
,
dict
):
continue
src
=
str
(
item
.
get
(
"src"
,
""
)).
strip
().
replace
(
"
\\
"
,
"/"
)
description
=
str
(
item
.
get
(
"description"
,
""
)).
strip
()
if
not
src
or
not
description
:
continue
if
src
.
startswith
(
"/"
)
or
Path
(
src
).
is_absolute
():
continue
normalized_path
=
Path
(
src
)
if
any
(
part
in
{
""
,
"."
,
".."
}
for
part
in
normalized_path
.
parts
):
continue
encoded_src
=
"/"
.
join
(
quote
(
part
)
for
part
in
normalized_path
.
parts
)
normalized_images
.
append
(
{
"src"
:
f
"
{
TASK_ASSET_URL_PREFIX
}
/
{
encoded_src
}
"
,
"description"
:
description
,
}
)
return
normalized_images
def
_build_task_payload
(
def
_build_task_payload
(
...
...
math-tutor/backend/scripts/convert_tasks_json_to_yaml.py
0 → 100644
View file @
5dac10a7
from
__future__
import
annotations
import
argparse
import
json
import
shutil
import
unicodedata
from
pathlib
import
Path
from
typing
import
Any
import
yaml
BACKEND_ROOT
=
Path
(
__file__
).
resolve
().
parents
[
1
]
TASKS_DIR
=
BACKEND_ROOT
/
"sources"
/
"tasks"
TASK_IMAGES_DIR
=
TASKS_DIR
/
"images"
DEFAULT_LEVEL
=
"Uebungen"
def
slugify
(
value
:
str
)
->
str
:
normalized
=
unicodedata
.
normalize
(
"NFKD"
,
str
(
value
))
ascii_value
=
normalized
.
encode
(
"ascii"
,
"ignore"
).
decode
(
"ascii"
)
lowered
=
ascii_value
.
strip
().
lower
()
collapsed
=
[]
last_was_dash
=
False
for
char
in
lowered
:
if
char
.
isalnum
():
collapsed
.
append
(
char
)
last_was_dash
=
False
elif
not
last_was_dash
:
collapsed
.
append
(
"-"
)
last_was_dash
=
True
return
""
.
join
(
collapsed
).
strip
(
"-"
)
def
load_json_task
(
path
:
Path
)
->
dict
[
str
,
Any
]:
payload
=
json
.
loads
(
path
.
read_text
(
encoding
=
"utf-8"
))
if
not
isinstance
(
payload
,
dict
):
raise
ValueError
(
f
"
{
path
.
name
}
: expected object"
)
if
not
isinstance
(
payload
.
get
(
"tasks"
),
list
):
raise
ValueError
(
f
"
{
path
.
name
}
: expected tasks list"
)
return
payload
def
_build_topic_entries
(
subsections
:
Any
)
->
list
[
dict
[
str
,
str
]]:
if
not
isinstance
(
subsections
,
list
):
return
[]
topics
:
list
[
dict
[
str
,
str
]]
=
[]
for
item
in
subsections
:
topic_id
=
str
(
item
).
strip
()
if
topic_id
:
topics
.
append
({
"topic-id"
:
topic_id
})
return
topics
def
_build_text_block
(
text
:
str
)
->
dict
[
str
,
str
]:
return
{
"type"
:
"text"
,
"text"
:
text
}
def
_build_image_block
(
src
:
str
,
description
:
str
)
->
dict
[
str
,
str
]:
return
{
"type"
:
"image"
,
"src"
:
src
,
"alt"
:
description
}
def
convert_task_payload
(
payload
:
dict
[
str
,
Any
],
source_file_id
:
str
,
image_path_map
:
dict
[
str
,
str
])
->
dict
[
str
,
Any
]:
title
=
str
(
payload
.
get
(
"title"
,
""
)).
strip
()
intro
=
str
(
payload
.
get
(
"intro"
,
""
)).
strip
()
slug
=
slugify
(
title
or
source_file_id
)
converted_tasks
:
list
[
dict
[
str
,
Any
]]
=
[]
for
raw_task
in
payload
.
get
(
"tasks"
,
[]):
if
not
isinstance
(
raw_task
,
dict
):
continue
aufgabe
:
list
[
dict
[
str
,
str
]]
=
[]
statement
=
str
(
raw_task
.
get
(
"statement"
,
""
)).
strip
()
if
statement
:
aufgabe
.
append
(
_build_text_block
(
statement
))
for
raw_image
in
raw_task
.
get
(
"images"
,
[])
if
isinstance
(
raw_task
.
get
(
"images"
),
list
)
else
[]:
if
not
isinstance
(
raw_image
,
dict
):
continue
old_src
=
str
(
raw_image
.
get
(
"src"
,
""
)).
strip
().
replace
(
"
\\
"
,
"/"
)
description
=
str
(
raw_image
.
get
(
"description"
,
""
)).
strip
()
new_src
=
image_path_map
.
get
(
old_src
,
old_src
)
if
old_src
and
description
:
aufgabe
.
append
(
_build_image_block
(
new_src
,
description
))
hinweise
:
list
[
dict
[
str
,
str
]]
=
[]
for
hint
in
raw_task
.
get
(
"hints"
,
[])
if
isinstance
(
raw_task
.
get
(
"hints"
),
list
)
else
[]:
hint_text
=
str
(
hint
).
strip
()
if
hint_text
:
hinweise
.
append
(
_build_text_block
(
hint_text
))
loesung_text
=
str
(
raw_task
.
get
(
"solution"
,
""
)).
strip
()
loesung
=
[
_build_text_block
(
loesung_text
)]
if
loesung_text
else
[]
converted_tasks
.
append
(
{
"aufgabe"
:
aufgabe
,
"hinweise"
:
hinweise
,
"loesung"
:
loesung
,
}
)
return
{
"id"
:
source_file_id
,
"titel"
:
title
,
"level"
:
DEFAULT_LEVEL
,
"slug"
:
slug
,
"description"
:
intro
,
"exercise_topic"
:
_build_topic_entries
(
payload
.
get
(
"subsections"
,
[])),
"tasks"
:
converted_tasks
,
}
def
build_image_path_map
(
payload
:
dict
[
str
,
Any
],
slug
:
str
)
->
dict
[
str
,
str
]:
mapping
:
dict
[
str
,
str
]
=
{}
used_targets
:
set
[
str
]
=
set
()
for
raw_task
in
payload
.
get
(
"tasks"
,
[]):
if
not
isinstance
(
raw_task
,
dict
):
continue
for
raw_image
in
raw_task
.
get
(
"images"
,
[])
if
isinstance
(
raw_task
.
get
(
"images"
),
list
)
else
[]:
if
not
isinstance
(
raw_image
,
dict
):
continue
old_src
=
str
(
raw_image
.
get
(
"src"
,
""
)).
strip
().
replace
(
"
\\
"
,
"/"
)
if
not
old_src
:
continue
basename
=
Path
(
old_src
).
name
stem
=
Path
(
basename
).
stem
suffix
=
Path
(
basename
).
suffix
target
=
f
"
{
slug
}
/
{
basename
}
"
counter
=
2
while
target
in
used_targets
:
target
=
f
"
{
slug
}
/
{
stem
}
-
{
counter
}{
suffix
}
"
counter
+=
1
used_targets
.
add
(
target
)
mapping
[
old_src
]
=
target
return
mapping
def
filter_existing_image_path_map
(
image_path_map
:
dict
[
str
,
str
],
images_root
:
Path
)
->
tuple
[
dict
[
str
,
str
],
list
[
str
]]:
existing
:
dict
[
str
,
str
]
=
{}
missing
:
list
[
str
]
=
[]
for
old_src
,
new_src
in
image_path_map
.
items
():
source_path
=
images_root
/
old_src
if
source_path
.
exists
():
existing
[
old_src
]
=
new_src
else
:
missing
.
append
(
old_src
)
return
existing
,
missing
def
move_task_images
(
image_path_map
:
dict
[
str
,
str
],
images_root
:
Path
,
dry_run
:
bool
=
False
)
->
list
[
str
]:
moved
:
list
[
str
]
=
[]
for
old_src
,
new_src
in
image_path_map
.
items
():
source_path
=
images_root
/
old_src
target_path
=
images_root
/
new_src
if
source_path
.
resolve
()
==
target_path
.
resolve
():
continue
if
dry_run
:
moved
.
append
(
f
"
{
source_path
}
->
{
target_path
}
"
)
continue
target_path
.
parent
.
mkdir
(
parents
=
True
,
exist_ok
=
True
)
shutil
.
move
(
str
(
source_path
),
str
(
target_path
))
moved
.
append
(
f
"
{
source_path
}
->
{
target_path
}
"
)
return
moved
def
remove_empty_dirs
(
root
:
Path
)
->
None
:
for
path
in
sorted
(
root
.
rglob
(
"*"
),
reverse
=
True
):
if
path
.
is_dir
():
try
:
next
(
path
.
iterdir
())
except
StopIteration
:
path
.
rmdir
()
def
build_output_yaml_path
(
tasks_dir
:
Path
,
slug
:
str
,
source_file_id
:
str
,
used_names
:
set
[
str
])
->
Path
:
candidates
=
[
slug
,
slugify
(
source_file_id
)]
for
candidate
in
candidates
:
if
candidate
and
candidate
not
in
used_names
:
used_names
.
add
(
candidate
)
return
tasks_dir
/
f
"
{
candidate
}
.yaml"
base_name
=
slug
or
slugify
(
source_file_id
)
or
"task"
counter
=
2
candidate
=
f
"
{
base_name
}
-
{
counter
}
"
while
candidate
in
used_names
:
counter
+=
1
candidate
=
f
"
{
base_name
}
-
{
counter
}
"
used_names
.
add
(
candidate
)
return
tasks_dir
/
f
"
{
candidate
}
.yaml"
def
convert_file
(
json_path
:
Path
,
output_path
:
Path
|
None
=
None
,
images_root
:
Path
|
None
=
None
,
dry_run
:
bool
=
False
,
)
->
tuple
[
Path
,
list
[
str
],
list
[
str
]]:
payload
=
load_json_task
(
json_path
)
source_file_id
=
json_path
.
stem
slug
=
slugify
(
str
(
payload
.
get
(
"title"
,
""
)).
strip
()
or
source_file_id
)
yaml_path
=
output_path
or
json_path
.
with_name
(
f
"
{
slug
}
.yaml"
)
resolved_images_root
=
images_root
or
TASK_IMAGES_DIR
image_path_map
=
build_image_path_map
(
payload
,
slug
)
existing_image_map
,
missing_images
=
filter_existing_image_path_map
(
image_path_map
,
resolved_images_root
)
converted
=
convert_task_payload
(
payload
,
source_file_id
,
existing_image_map
)
moves
=
move_task_images
(
existing_image_map
,
resolved_images_root
,
dry_run
=
dry_run
)
if
not
dry_run
:
yaml_path
.
write_text
(
yaml
.
safe_dump
(
converted
,
sort_keys
=
False
,
allow_unicode
=
True
),
encoding
=
"utf-8"
,
)
json_path
.
unlink
()
return
yaml_path
,
moves
,
missing_images
def
convert_all
(
tasks_dir
:
Path
=
TASKS_DIR
,
dry_run
:
bool
=
False
)
->
tuple
[
list
[
Path
],
list
[
str
]]:
converted_paths
:
list
[
Path
]
=
[]
missing_images
:
list
[
str
]
=
[]
images_root
=
tasks_dir
/
"images"
used_names
:
set
[
str
]
=
set
()
for
json_path
in
sorted
(
tasks_dir
.
glob
(
"*.json"
)):
if
json_path
.
name
==
"_subsection_map.json"
:
continue
payload
=
load_json_task
(
json_path
)
source_file_id
=
json_path
.
stem
slug
=
slugify
(
str
(
payload
.
get
(
"title"
,
""
)).
strip
()
or
source_file_id
)
yaml_path
=
build_output_yaml_path
(
tasks_dir
,
slug
,
source_file_id
,
used_names
)
yaml_path
,
_
,
missing
=
convert_file
(
json_path
,
output_path
=
yaml_path
,
images_root
=
images_root
,
dry_run
=
dry_run
,
)
converted_paths
.
append
(
yaml_path
)
missing_images
.
extend
(
missing
)
if
not
dry_run
and
images_root
.
exists
():
remove_empty_dirs
(
images_root
)
return
converted_paths
,
missing_images
def
main
()
->
int
:
parser
=
argparse
.
ArgumentParser
(
description
=
"Convert backend task sources from JSON to YAML."
)
parser
.
add_argument
(
"--dry-run"
,
action
=
"store_true"
,
help
=
"Preview conversion without writing files."
)
args
=
parser
.
parse_args
()
converted_paths
,
missing_images
=
convert_all
(
dry_run
=
args
.
dry_run
)
print
(
f
"Converted
{
len
(
converted_paths
)
}
task files."
)
if
missing_images
:
print
(
f
"Missing images preserved with original src (
{
len
(
missing_images
)
}
):"
)
for
item
in
missing_images
:
print
(
f
"-
{
item
}
"
)
return
0
if
__name__
==
"__main__"
:
raise
SystemExit
(
main
())
math-tutor/backend/test/task_yaml_migration_test.py
0 → 100644
View file @
5dac10a7
from
__future__
import
annotations
import
json
import
os
import
shutil
import
tempfile
import
unittest
from
pathlib
import
Path
import
yaml
os
.
environ
.
setdefault
(
"OPENAI_BASE_URL"
,
"http://localhost:9999"
)
os
.
environ
.
setdefault
(
"OPENAI_API_KEY"
,
"test-key"
)
os
.
environ
.
setdefault
(
"POSTGRES_URL"
,
"postgresql://localhost/test"
)
from
app.deterministic_services
import
task_catalog
from
scripts
import
convert_tasks_json_to_yaml
class
TaskYamlMigrationTest
(
unittest
.
TestCase
):
def
test_load_task_files_normalizes_yaml_schema
(
self
)
->
None
:
temp_dir
=
Path
(
tempfile
.
mkdtemp
(
prefix
=
"task-yaml-"
))
yaml_path
=
temp_dir
/
"analysis.yaml"
yaml_path
.
write_text
(
yaml
.
safe_dump
(
{
"id"
:
"analysis_1"
,
"titel"
:
"Analysis"
,
"level"
:
"Uebungen"
,
"slug"
:
"analysis"
,
"description"
:
"Intro"
,
"exercise_topic"
:
[{
"topic-id"
:
"quadratische-gleichungen"
}],
"tasks"
:
[
{
"aufgabe"
:
[
{
"type"
:
"text"
,
"text"
:
"Bestimme f(x)."
},
{
"type"
:
"image"
,
"src"
:
"analysis/01.png"
,
"alt"
:
"Graph einer Parabel."
},
],
"hinweise"
:
[
{
"type"
:
"text"
,
"text"
:
"Nutze die Ableitung."
},
{
"type"
:
"image"
,
"src"
:
"analysis/h1.png"
,
"alt"
:
"Hinweisgrafik."
},
],
"loesung"
:
[{
"type"
:
"text"
,
"text"
:
"f'(x)=2x"
}],
}
],
},
sort_keys
=
False
,
allow_unicode
=
True
,
),
encoding
=
"utf-8"
,
)
try
:
loaded
=
task_catalog
.
load_task_files
(
temp_dir
)
finally
:
shutil
.
rmtree
(
temp_dir
,
ignore_errors
=
True
)
self
.
assertEqual
(
len
(
loaded
),
1
)
self
.
assertEqual
(
loaded
[
0
][
"_file_id"
],
"analysis_1"
)
self
.
assertEqual
(
loaded
[
0
][
"title"
],
"Analysis"
)
self
.
assertEqual
(
loaded
[
0
][
"intro"
],
"Intro"
)
self
.
assertEqual
(
loaded
[
0
][
"subsections"
],
[
"quadratische-gleichungen"
])
self
.
assertEqual
(
loaded
[
0
][
"tasks"
][
0
][
"id"
],
"01"
)
self
.
assertEqual
(
loaded
[
0
][
"tasks"
][
0
][
"statement"
],
"Bestimme f(x)."
)
self
.
assertEqual
(
loaded
[
0
][
"tasks"
][
0
][
"hints"
],
[
"Nutze die Ableitung."
])
self
.
assertEqual
(
loaded
[
0
][
"tasks"
][
0
][
"solution"
],
"f'(x)=2x"
)
self
.
assertEqual
(
loaded
[
0
][
"tasks"
][
0
][
"images"
],
[
{
"src"
:
"analysis/01.png"
,
"description"
:
"Graph einer Parabel."
},
{
"src"
:
"analysis/h1.png"
,
"description"
:
"Hinweisgrafik."
},
],
)
def
test_load_task_files_skips_invalid_yaml
(
self
)
->
None
:
temp_dir
=
Path
(
tempfile
.
mkdtemp
(
prefix
=
"task-yaml-invalid-"
))
valid_path
=
temp_dir
/
"valid.yaml"
invalid_path
=
temp_dir
/
"invalid.yaml"
malformed_path
=
temp_dir
/
"malformed.yaml"
valid_path
.
write_text
(
yaml
.
safe_dump
(
{
"id"
:
"valid"
,
"titel"
:
"Valid"
,
"level"
:
"Uebungen"
,
"slug"
:
"valid"
,
"description"
:
"Intro"
,
"exercise_topic"
:
[],
"tasks"
:
[{
"aufgabe"
:
[],
"hinweise"
:
[],
"loesung"
:
[]}],
},
sort_keys
=
False
,
allow_unicode
=
True
,
),
encoding
=
"utf-8"
,
)
invalid_path
.
write_text
(
yaml
.
safe_dump
(
{
"id"
:
"invalid"
,
"titel"
:
"Invalid"
,
"slug"
:
"invalid"
,
"description"
:
"Intro"
,
"exercise_topic"
:
[],
"tasks"
:
[],
},
sort_keys
=
False
,
allow_unicode
=
True
,
),
encoding
=
"utf-8"
,
)
malformed_path
.
write_text
(
"tasks: [
\n
"
,
encoding
=
"utf-8"
)
try
:
loaded
=
task_catalog
.
load_task_files
(
temp_dir
)
finally
:
shutil
.
rmtree
(
temp_dir
,
ignore_errors
=
True
)
self
.
assertEqual
([
item
[
"_file_id"
]
for
item
in
loaded
],
[
"valid"
])
def
test_build_task_catalog_keeps_api_shape_for_yaml_sources
(
self
)
->
None
:
task_files
=
[
{
"_file_id"
:
"analysis_1"
,
"title"
:
"Analysis"
,
"intro"
:
"Intro"
,
"subsections"
:
[
"quadratische-gleichungen"
],
"tasks"
:
[
{
"id"
:
"01"
,
"statement"
:
"Bestimme f(x)."
,
"hints"
:
[],
"solution"
:
"f'(x)=2x"
,
"images"
:
[{
"src"
:
"analysis/01.png"
,
"description"
:
"Graph einer Parabel."
}],
}
],
}
]
catalog
=
task_catalog
.
build_task_catalog
(
task_files
)
self
.
assertEqual
(
catalog
[
0
][
"file_id"
],
"analysis_1"
)
self
.
assertEqual
(
catalog
[
0
][
"title"
],
"Analysis"
)
self
.
assertEqual
(
catalog
[
0
][
"intro"
],
"Intro"
)
self
.
assertEqual
(
catalog
[
0
][
"tasks"
][
0
][
"task_id"
],
"01"
)
self
.
assertEqual
(
catalog
[
0
][
"tasks"
][
0
][
"statement"
],
"Bestimme f(x)."
)
self
.
assertEqual
(
catalog
[
0
][
"tasks"
][
0
][
"images"
],
[{
"src"
:
"/api/tasks/assets/analysis/01.png"
,
"description"
:
"Graph einer Parabel."
}],
)
def
test_convert_task_payload_maps_json_shape_to_yaml
(
self
)
->
None
:
payload
=
{
"title"
:
"Reelle Funktionen bestimmen"
,
"subsections"
:
[
"eigenschaften-reeller-funktionen"
],
"intro"
:
"Geben Sie Definitions- und Wertebereich der Funktion an:"
,
"tasks"
:
[
{
"id"
:
"01"
,
"statement"
:
"$f(x)=x^2-3$"
,
"images"
:
[{
"src"
:
"reelle_funktionen_bestimmen/Bild1.png"
,
"description"
:
"Graph"
}],
"hints"
:
[
"Hinweis 1"
,
""
],
"solution"
:
"Loesung"
,
}
],
}
converted
=
convert_tasks_json_to_yaml
.
convert_task_payload
(
payload
,
source_file_id
=
"Reelle Funktionen bestimmen"
,
image_path_map
=
{
"reelle_funktionen_bestimmen/Bild1.png"
:
"reelle-funktionen-bestimmen/Bild1.png"
},
)
self
.
assertEqual
(
converted
[
"id"
],
"Reelle Funktionen bestimmen"
)
self
.
assertEqual
(
converted
[
"slug"
],
"reelle-funktionen-bestimmen"
)
self
.
assertEqual
(
converted
[
"level"
],
"Uebungen"
)
self
.
assertEqual
(
converted
[
"exercise_topic"
],
[{
"topic-id"
:
"eigenschaften-reeller-funktionen"
}])
self
.
assertEqual
(
converted
[
"tasks"
][
0
][
"aufgabe"
][
0
],
{
"type"
:
"text"
,
"text"
:
"$f(x)=x^2-3$"
})
self
.
assertEqual
(
converted
[
"tasks"
][
0
][
"aufgabe"
][
1
],
{
"type"
:
"image"
,
"src"
:
"reelle-funktionen-bestimmen/Bild1.png"
,
"alt"
:
"Graph"
},
)
self
.
assertEqual
(
converted
[
"tasks"
][
0
][
"hinweise"
],
[{
"type"
:
"text"
,
"text"
:
"Hinweis 1"
}])
self
.
assertEqual
(
converted
[
"tasks"
][
0
][
"loesung"
],
[{
"type"
:
"text"
,
"text"
:
"Loesung"
}])
def
test_convert_all_rewrites_json_and_moves_images
(
self
)
->
None
:
temp_dir
=
Path
(
tempfile
.
mkdtemp
(
prefix
=
"task-convert-"
))
images_dir
=
temp_dir
/
"images"
/
"old-folder"
images_dir
.
mkdir
(
parents
=
True
,
exist_ok
=
True
)
(
images_dir
/
"Bild1.png"
).
write_text
(
"image"
,
encoding
=
"utf-8"
)
json_path
=
temp_dir
/
"Reelle Funktionen bestimmen.json"
json_path
.
write_text
(
json
.
dumps
(
{
"title"
:
"Reelle Funktionen bestimmen"
,
"subsections"
:
[
"eigenschaften-reeller-funktionen"
],
"intro"
:
"Intro"
,
"tasks"
:
[
{
"id"
:
"01"
,
"statement"
:
"$f(x)=x^2-3$"
,
"images"
:
[{
"src"
:
"old-folder/Bild1.png"
,
"description"
:
"Graph"
}],
"hints"
:
[
"Hinweis 1"
],
"solution"
:
"Loesung"
,
}
],
},
ensure_ascii
=
False
,
),
encoding
=
"utf-8"
,
)
try
:
converted
,
missing_images
=
convert_tasks_json_to_yaml
.
convert_all
(
temp_dir
)
yaml_path
=
temp_dir
/
"reelle-funktionen-bestimmen.yaml"
self
.
assertEqual
(
converted
,
[
yaml_path
])
self
.
assertEqual
(
missing_images
,
[])
self
.
assertFalse
(
json_path
.
exists
())
self
.
assertTrue
(
yaml_path
.
exists
())
self
.
assertTrue
((
temp_dir
/
"images"
/
"reelle-funktionen-bestimmen"
/
"Bild1.png"
).
exists
())
payload
=
yaml
.
safe_load
(
yaml_path
.
read_text
(
encoding
=
"utf-8"
))
self
.
assertEqual
(
payload
[
"id"
],
"Reelle Funktionen bestimmen"
)
self
.
assertEqual
(
payload
[
"tasks"
][
0
][
"aufgabe"
][
1
][
"src"
],
"reelle-funktionen-bestimmen/Bild1.png"
,
)
finally
:
shutil
.
rmtree
(
temp_dir
,
ignore_errors
=
True
)
def
test_convert_all_uses_distinct_yaml_filenames_for_duplicate_titles
(
self
)
->
None
:
temp_dir
=
Path
(
tempfile
.
mkdtemp
(
prefix
=
"task-convert-duplicate-"
))
first_json
=
temp_dir
/
"Auf Beschränktheit untersuchen.json"
second_json
=
temp_dir
/
"Auf Beschränktheit untersuchen_images.json"
payload
=
{
"title"
:
"Auf Beschränktheit untersuchen"
,
"subsections"
:
[],
"intro"
:
"Intro"
,
"tasks"
:
[{
"id"
:
"01"
,
"statement"
:
"s"
,
"hints"
:
[],
"solution"
:
"l"
}],
}
first_json
.
write_text
(
json
.
dumps
(
payload
,
ensure_ascii
=
False
),
encoding
=
"utf-8"
)
second_json
.
write_text
(
json
.
dumps
(
payload
,
ensure_ascii
=
False
),
encoding
=
"utf-8"
)
try
:
converted
,
missing_images
=
convert_tasks_json_to_yaml
.
convert_all
(
temp_dir
)
self
.
assertEqual
(
missing_images
,
[])
self
.
assertEqual
(
sorted
(
path
.
name
for
path
in
converted
),
[
"auf-beschranktheit-untersuchen-images.yaml"
,
"auf-beschranktheit-untersuchen.yaml"
,
],
)
finally
:
shutil
.
rmtree
(
temp_dir
,
ignore_errors
=
True
)
if
__name__
==
"__main__"
:
unittest
.
main
()
Write
Preview
Supports
Markdown
0%
Try again
or
attach a new file
.
Cancel
You are about to add
0
people
to the discussion. Proceed with caution.
Finish editing this message first!
Cancel
Please
register
or
sign in
to comment