fix(files): detect CSV dialect before parsing

This commit is contained in:
Mikei386
2026-08-24 07:05:04 +02:00
parent e24839bfe1
commit c3121281de
4 changed files with 16 additions and 4 deletions
+2
View File
@@ -139,6 +139,8 @@ class StabilityGuardTests(unittest.IsolatedAsyncioTestCase):
["execute_code"],
)
self.assertIn("private table rule", result["messages"][0]["content"])
self.assertIn("sep=None", result["messages"][0]["content"])
self.assertIn("delimiter", result["messages"][0]["content"])
async def test_private_csv_is_detected_from_user_text_without_metadata(self):
body = {
+4 -1
View File
@@ -22,7 +22,10 @@ benötigte deshalb kleinere, klarere Werkzeuge und harte Abbruchgrenzen.
erhalten keine Dateiinhalte oder daraus abgeleitete Suchbegriffe. Der Filter
leert dafür die MCP-Auswahl und deaktiviert `features.web_search`; im
installierten OpenWebUI-Code läuft der Filter nachweislich vor der
Webwerkzeug-Injektion.
Webwerkzeug-Injektion. Vor `pandas.read_csv` werden Rohvorschau, Kodierung,
Trennzeichen, Kopfzeile, Metadatenzeilen, Dezimal- und Datumsformat erkannt;
damit führen deutsche Bankexporte nicht mehr unnötig zuerst zu einem
ParserError wegen einer falschen Spaltenzahl.
4. Pro Antwort sind höchstens zwölf Werkzeugrunden erlaubt. Der zweite
identische Aufruf wird gestoppt. Ein einzelnes Resultat ist auf 10.000, alle
Resultate zusammen auf 36.000 Zeichen begrenzt.
@@ -143,7 +143,11 @@ class Filter:
"code interpreter using pandas/openpyxl. Never send its filename, contents, "
"values, account data, categories or derived search terms to web or MCP "
"tools. Do not use Knowledge/RAG search to calculate totals. First inspect "
"columns and numeric/date formats, then compute exact aggregates, validate "
"a small raw preview and detect encoding, delimiter, quote rules, metadata "
"lines, header row, decimal separator and date format before parsing. Never "
"start with a blind pandas.read_csv(path) call. Prefer csv.Sniffer or "
"pandas.read_csv(path, sep=None, engine='python') for discovery, then parse "
"again with explicit verified parameters. Compute exact aggregates, validate "
"that totals reconcile, and always return a visible final answer or a clear "
"local parsing error."
)
+5 -2
View File
@@ -202,8 +202,11 @@ params = {
"Python code interpreter with pandas/openpyxl. Never send private file names, "
"contents, values, account data, categories, or derived search terms to any "
"web or MCP tool. Do not use Knowledge/RAG retrieval to calculate table totals. "
"Inspect the columns and locale-specific number/date formats, compute exact "
"aggregates, reconcile the result, and always provide a visible final answer. "
"Inspect a small raw preview first. Detect encoding, delimiter, quoting, metadata "
"lines, header row, decimal separator and date format before parsing. Do not begin "
"with a blind pandas.read_csv(path) call; use csv.Sniffer or sep=None with the "
"Python engine for discovery, then parse with explicit verified parameters. "
"Compute exact aggregates, reconcile the result, and always provide a visible final answer. "
"For GitHub repository implementation details, README files, source trees, "
"API routes, or code search, use the dedicated official GitHub repository "
"tool instead of guessing from ordinary web results. Use general web search "