Commit b6dd9af8 authored by Samuel Maier's avatar Samuel Maier
Browse files

Update snapshot

parents
"""
Contains functions to actually parse moodles xml question format (using `moodle_questions_dataclasses.py`).
If executed as main it tests these methods, and also determines some statistics.
"""
from xml.etree import ElementTree
from common_py.utils import UNREACHABLE
from typing import Iterable, Tuple
from itertools import product
import moodle_questions_dataclasses as mqd
class CollectDebugInfos():
"Just a little record keeping for debugging purposes"
questionTypesMetaInfo: dict
def __init__(self):
self.questionTypesMetaInfo = {}
def collectData(self, question: ElementTree.Element):
questionType = question.attrib["type"]
if questionType not in self.questionTypesMetaInfo:
self.questionTypesMetaInfo[questionType] = {
"child_attrib": {},
}
debug_meta_info = self.questionTypesMetaInfo[questionType]
for item in question.findall("./"):
if item.tag not in debug_meta_info["child_attrib"]:
debug_meta_info["child_attrib"][item.tag] = {}
def matchAndCreateQuestion(question: ElementTree.Element, schoolClassAsked: str | None):
common = mqd.CommonQuestionFields.createFromXml(question, schoolClassAsked)
match question.attrib["type"]:
case "multichoice":
return mqd.MultiChoice.createFromXml(question, common)
case "truefalse":
return mqd.TrueFalse.createFromXml(question, common)
case "matching":
return mqd.Matching.createFromXml(question, common)
case "ddimageortext":
return mqd.NotHandled.createFromXml(question, common=common, reason="Very likely to contain/require images")
# These questions are only present in PGM1+2
case "cloze":
return mqd.NotHandled.createFromXml(question, common=common, reason="not a closed question")
case "gapselect":
return mqd.Gapselect.createFromXml(question, common)
# "Move item to the right spot in the sentence", the right choice is encoded in the marker, [[i]] means the ith answer (beginning with 1) is the correct choice.
case "ddwtos":
return mqd.Ddwtos.createFromXml(question, common)
case _:
UNREACHABLE("Handle all possible question types")
def extractQuestionsFromMoodleQuizXml(
subjectAndXmlPaths: Iterable[Tuple[str, str]]
) -> mqd.QuestionsDataType:
":param subjectAndXmlPaths: Iterator of Tuple Pairs of (subject, xml_path)"
questionsData: mqd.QuestionsDataType = []
debugInfo = CollectDebugInfos()
for (subject, xmlPath) in subjectAndXmlPaths:
for question in ElementTree.parse(xmlPath).getroot().iter("question"):
debugInfo.collectData(question)
questionType = question.attrib["type"]
# "category" questions are not actually questions it appears
if not question.find("name"):
assert questionType == "category"
continue
# adding a bunch of assertions to avoid silent failure (also in following match)
assert question.find("./questiontext").attrib["format"] == "html"
questionsData.append(matchAndCreateQuestion(question, schoolClassAsked=subject))
return questionsData
if __name__ == "__main__":
subjects = ["ASV", "KI", "MLDM", "PGM_1_2"]
questionsData = extractQuestionsFromMoodleQuizXml(
map(lambda subject: (subject, f"../raw_data/hft/{subject}/quiz.moodle.xml"), subjects)
)
# pprint(questionsData)
questionTypes = [
"multichoice",
"truefalse",
"matching",
"ddimageortext",
"cloze",
"gapselect",
"ddwtos",
]
print("Table")
for (maybeSubject, maybeType) in product([*subjects, None], [*questionTypes, None]):
print(f"\t{maybeSubject}, {maybeType}: {len([itm for itm in questionsData if (not maybeSubject or itm.common.schoolClassAsked == maybeSubject) and (not maybeType or itm.common.questionType == maybeType)])}\n")
questionsWithoutFiles = [itm for itm in questionsData if not itm.common.questionContainsFile]
print("Table")
for (maybeSubject, maybeType) in product([*subjects, None], [*questionTypes, None]): #"singlechoice",
print(f"\t{maybeSubject}, {maybeType}: {len([itm for itm in questionsWithoutFiles if (not maybeSubject or itm.common.schoolClassAsked == maybeSubject) and (not maybeType or (itm.common.questionType == maybeType))])}\n") #and (not maybeType.endswith('choice') or (itm.singleChoice if maybeType == 'singlechoice' else not itm.singleChoice))
if maybeType and maybeType.endswith("choice"):
print(f"single: {len([itm for itm in questionsWithoutFiles if (not maybeSubject or itm.common.schoolClassAsked == maybeSubject) and (not maybeType or (itm.common.questionType == maybeType and itm.singleChoice))])}")
print(f"multi: {len([itm for itm in questionsWithoutFiles if (not maybeSubject or itm.common.schoolClassAsked == maybeSubject) and (not maybeType or (itm.common.questionType == maybeType and not itm.singleChoice))])}")
# countAll = 0
# countTrueFalse = 0
# for question in questionsData:
# countAll += 1
# if isinstance(question, TrueFalse):
# countTrueFalse += 1
# print(countAll, countTrueFalse)
for questionData in questionsData:
match questionData:
case mqd.Matching(common=mqd.CommonQuestionFields(questionText = questionText)):
pass
# UNREACHABLE("testing matching")
"""
This file contains fairly simple logic that converts generic
questions to answers as they were found in the enem dataset.
"""
from common_py.string_processing import cleanHtml, stripLines, combineStrIter
from moodle_to_generic_questions import GenericQuestion, AnswerOption
def mapGenericQuestionToAnswerStr(input: GenericQuestion):
return stripLines(combineStrIter([f"#{'T' if ans['correct'] else 'F'} {cleanHtml(ans['text'])}" for ans in input.answerOptions]))
def mapGenericQuestionToQuestionStr(input: GenericQuestion):
return stripLines(cleanHtml(input.questionText))
def mapGenericQuestionToStr(input: GenericQuestion):
return f"{mapGenericQuestionToQuestionStr(input)}{mapGenericQuestionToAnswerStr(input)}"
if __name__=="__main__":
print(
mapGenericQuestionToStr(GenericQuestion("alaaa", "<h>headertext</h>", [AnswerOption(text="Answertext", correct=False)], "whatever"))
)
\ No newline at end of file
This diff is collapsed.
"""
Contains functions to convert the multliple variants of moodle question types to a more generic one.
This makes a few choices in how these are mapped!
"""
from dataclasses import dataclass
import moodle_questions_dataclasses as mdt
from itertools import product
from common_py.utils import UNREACHABLE
from pprint import pprint
from typing import TypedDict
# polars doesnt like lists of dataclasses as elements,
# but for earlier processing TypedDict is uncomfortable,
# as one cant do type-based match on that
# So I end up using both, oh well.
class AnswerOption(TypedDict):
text: str
"Enriched with HTML"
# recievedPointsFract: float
correct: bool
@dataclass
class GenericQuestion():
name: str
"This is the name from the moodle XML export, not the csv"
questionText: str
"Enriched with HTML"
answerOptions: list[AnswerOption]
questionType: str
"The original question type (to allow filtering in case some types are suspected to be problematic)"
def mapToGenericQuestionFormat(
input: mdt.QuestionsDataType,
):
trueMultichoiceOptions = []
matchingOptions = []
skippedBecauseImage = []
skippedForAnotherReason = []
output: list[GenericQuestion] = []
for elem in input:
meta = elem.common
match elem:
case mdt.NotHandled(name, _):
skippedForAnotherReason.append(name)
# Skip questions which contain an image (assumes that all types contain meta as first field)
case elem if elem.common.questionContainsFile:
skippedBecauseImage +=[meta.name]
continue
case mdt.MultiChoice(answers = answersSrc, singleChoice = singleChoice):
answers: list[AnswerOption] = []
if singleChoice:
answers.extend([
AnswerOption(
text = ans.text,
# recievedPointsFract=0.0 if ans.correctFract is None else 1.0,
correct=False if ans.correctFract is None else True,
) for ans in answersSrc
])
else:
trueMultichoiceOptions += [len(answersSrc)]
selections = answersSrc
answers.extend([
AnswerOption(
text = ans.text,
correct = False if ans.correctFract is None else True,
# recievedPointsFract=0,
) for ans in selections
])
output.append(GenericQuestion(
name=meta.name,
questionText=meta.questionText,
answerOptions=answers,
questionType= "singlechoice" if singleChoice else meta.questionType,
))
case mdt.TrueFalse(answers = answersSrc):
answers = [
AnswerOption(
text = "Die Aussage stimmt" if ans.correct else "Die Aussage stimmt nicht",
# recievedPointsFract=1.0 if ans.correct else 0.0,
correct = ans.correct,
) for ans in answersSrc
]
output.append(GenericQuestion(
name=meta.name,
questionText=meta.questionText,
answerOptions=answers,
questionType=meta.questionType,
))
case mdt.Matching(subQuestions = subQuestionsSrc):
answers: list[AnswerOption] = []
for ((subqIdx, subquestion), (ansIdx, answer)) in product(
enumerate([ans.text for ans in subQuestionsSrc]),
enumerate([ans.answer for ans in subQuestionsSrc])
):
answers.append(AnswerOption(
text = f"{answer} passt zu {subquestion}",
correct = subqIdx == ansIdx,
))
# for matchedIndexes in permutations(range(len(subQuestionsSrc))):
# answerTextParts = [
# f"{subQuestionsSrc[answerIdx].answer} passt zu {subQuestionsSrc[textIdx].text}"
# for (textIdx, answerIdx) in enumerate(matchedIndexes)
# ]
# answerPoints = meta.defaultgrade - sum([
# textIdx != answerIdx
# for (textIdx, answerIdx) in enumerate(matchedIndexes)
# ]) * meta.penalty
# answers.append(AnswerOption(
# text = reduce(lambda rhs, lhs: f"{rhs}, {lhs}", answerTextParts[1:], answerTextParts[0]),
# recievedPointsFract=float(answerPoints) / meta.defaultgrade,
# ))
matchingOptions += [len(answers)]
output.append(GenericQuestion(
name = meta.name,
questionText=meta.questionText,
answerOptions= answers,
questionType=meta.questionType,
))
case mdt.Ddwtos(dragboxes=dragBoxesSrc, common=mdt.CommonQuestionFields(name, _)):
skippedForAnotherReason.append(name)
# I wont put in that work as this is unlikely to work,
# considering that the generated text will be too long
continue
case mdt.Gapselect(selectOptions = selectOptionsSrc, common=mdt.CommonQuestionFields(name, _)):
skippedForAnotherReason.append(name)
# I wont put in that work as this is unlikely to work,
# considering that the generated text will be too long
continue
case _:
UNREACHABLE("Please be exhaustive (explicitly skip ignored stuff)")
print(f"""
#options true multichoice" {trueMultichoiceOptions}
#options match (also true multichoice) {matchingOptions}
#skipped because of pictures {len(skippedBecauseImage)}
#skipped because of other reason {len(skippedForAnotherReason)}
""")
return output
if __name__ == "__main__":
from extract_moodle_xml import extractQuestionsFromMoodleQuizXml
pprint(
len(mapToGenericQuestionFormat(extractQuestionsFromMoodleQuizXml(
map(lambda subject: (subject, f"../raw_data/hft/{subject}/quiz.moodle.xml"), ["ASV", "KI", "MLDM", "PGM_1_2"])
)))
)
# FIXME: Determine typical context length for BERT etc, maybe some of the answer serializations are not possible as they result in too much text
# Alternative would be some other encoding, but its doubtful whether the LLM can learn that encoding with out dataset.
# EG keeping the [[1]] fields in [[2]] the sentence and providing an order for answers
# Is there some sparse encoding example in typical pretraining data/tasks?
# TrueFalse: Prepend with "Ist der foldende Satz wahr oder falsch?", use these as variants?
# Matching: subq.text " passt zu " subq.answer
# DDWTOS: "Welcher der folgenden sätze ist korrekt?", options with text filled in
# FIXME: Kein true multiple choice in ENEM (Regex /#T((.*?)\n #){1,8}T/, getested an einem gefaktem positive)
# FIXME: Enthält als einziges itemize|array|footnotesize umgebungen (keine tables)
\ No newline at end of file
This diff is collapsed.
This diff is collapsed.
results
__pycache__
.mypy_cache
.ruff_cache
hash\[*
\ No newline at end of file
This diff is collapsed.
This diff is collapsed.
This diff is collapsed.
This diff is collapsed.
This diff is collapsed.
This diff is collapsed.
This diff is collapsed.
This diff is collapsed.
#!/bin/sh
#SBATCH --partition=dev_gpu_4_a100
#SBATCH --output=pip_list_report.log
#SBATCH --gres=gpu:1
#SBATCH --cpus-per-task=1
enroot start -m "${PWD}:/workspace" nv_tensor_container_fresh sh <<EOF
pip list
EOF
This diff is collapsed.
This diff is collapsed.
This diff is collapsed.
This diff is collapsed.
Supports Markdown
0% or .
You are about to add 0 people to the discussion. Proceed with caution.
Finish editing this message first!
Please register or to comment