fix(files): detect CSV dialect before parsing
This commit is contained in:
@@ -139,6 +139,8 @@ class StabilityGuardTests(unittest.IsolatedAsyncioTestCase):
|
||||
["execute_code"],
|
||||
)
|
||||
self.assertIn("private table rule", result["messages"][0]["content"])
|
||||
self.assertIn("sep=None", result["messages"][0]["content"])
|
||||
self.assertIn("delimiter", result["messages"][0]["content"])
|
||||
|
||||
async def test_private_csv_is_detected_from_user_text_without_metadata(self):
|
||||
body = {
|
||||
|
||||
@@ -22,7 +22,10 @@ benötigte deshalb kleinere, klarere Werkzeuge und harte Abbruchgrenzen.
|
||||
erhalten keine Dateiinhalte oder daraus abgeleitete Suchbegriffe. Der Filter
|
||||
leert dafür die MCP-Auswahl und deaktiviert `features.web_search`; im
|
||||
installierten OpenWebUI-Code läuft der Filter nachweislich vor der
|
||||
Webwerkzeug-Injektion.
|
||||
Webwerkzeug-Injektion. Vor `pandas.read_csv` werden Rohvorschau, Kodierung,
|
||||
Trennzeichen, Kopfzeile, Metadatenzeilen, Dezimal- und Datumsformat erkannt;
|
||||
damit führen deutsche Bankexporte nicht mehr unnötig zuerst zu einem
|
||||
ParserError wegen einer falschen Spaltenzahl.
|
||||
4. Pro Antwort sind höchstens zwölf Werkzeugrunden erlaubt. Der zweite
|
||||
identische Aufruf wird gestoppt. Ein einzelnes Resultat ist auf 10.000, alle
|
||||
Resultate zusammen auf 36.000 Zeichen begrenzt.
|
||||
|
||||
@@ -143,7 +143,11 @@ class Filter:
|
||||
"code interpreter using pandas/openpyxl. Never send its filename, contents, "
|
||||
"values, account data, categories or derived search terms to web or MCP "
|
||||
"tools. Do not use Knowledge/RAG search to calculate totals. First inspect "
|
||||
"columns and numeric/date formats, then compute exact aggregates, validate "
|
||||
"a small raw preview and detect encoding, delimiter, quote rules, metadata "
|
||||
"lines, header row, decimal separator and date format before parsing. Never "
|
||||
"start with a blind pandas.read_csv(path) call. Prefer csv.Sniffer or "
|
||||
"pandas.read_csv(path, sep=None, engine='python') for discovery, then parse "
|
||||
"again with explicit verified parameters. Compute exact aggregates, validate "
|
||||
"that totals reconcile, and always return a visible final answer or a clear "
|
||||
"local parsing error."
|
||||
)
|
||||
|
||||
@@ -202,8 +202,11 @@ params = {
|
||||
"Python code interpreter with pandas/openpyxl. Never send private file names, "
|
||||
"contents, values, account data, categories, or derived search terms to any "
|
||||
"web or MCP tool. Do not use Knowledge/RAG retrieval to calculate table totals. "
|
||||
"Inspect the columns and locale-specific number/date formats, compute exact "
|
||||
"aggregates, reconcile the result, and always provide a visible final answer. "
|
||||
"Inspect a small raw preview first. Detect encoding, delimiter, quoting, metadata "
|
||||
"lines, header row, decimal separator and date format before parsing. Do not begin "
|
||||
"with a blind pandas.read_csv(path) call; use csv.Sniffer or sep=None with the "
|
||||
"Python engine for discovery, then parse with explicit verified parameters. "
|
||||
"Compute exact aggregates, reconcile the result, and always provide a visible final answer. "
|
||||
"For GitHub repository implementation details, README files, source trees, "
|
||||
"API routes, or code search, use the dedicated official GitHub repository "
|
||||
"tool instead of guessing from ordinary web results. Use general web search "
|
||||
|
||||
Reference in New Issue
Block a user