mirror of
https://github.com/supabase/supabase.git
synced 2026-10-06 09:55:06 +03:00
[Midway] Shift column inference logic for CSV import to CSV parsing step instead of saving CSV content step
This commit is contained in:
1 parent
3497398a1e
commit
4723b8c231
5 files changed
+83
-69
No files matched your search
+6
-5
@@ -33,6 +33,7 @@ const SpreadsheetImport: FC<Props> = ({
|
||||
headers: headers,
|
||||
rows: rows,
|
||||
rowCount: 0,
|
||||
columnTypeMap: {},
|
||||
})
|
||||
|
||||
const [input, setInput] = useState<string>('')
|
||||
@@ -56,7 +57,7 @@ const SpreadsheetImport: FC<Props> = ({
|
||||
})
|
||||
} else {
|
||||
setUploadedFile(file)
|
||||
const { headers, rowCount, rowPreview, errors } = await parseSpreadsheet(
|
||||
const { headers, rowCount, columnTypeMap, errors } = await parseSpreadsheet(
|
||||
file,
|
||||
onProgressUpdate
|
||||
)
|
||||
@@ -74,13 +75,13 @@ const SpreadsheetImport: FC<Props> = ({
|
||||
message: `Multiple errors have been detected on ${errors.length} rows. Do check the file you have uploaded for any discrepancies.`,
|
||||
})
|
||||
}
|
||||
setSpreadsheetData({ headers, rowCount, rows: rowPreview })
|
||||
setSpreadsheetData({ headers, rows: [], rowCount, columnTypeMap })
|
||||
}
|
||||
event.target.value = ''
|
||||
}
|
||||
|
||||
const removeUploadedFile = () => {
|
||||
setSpreadsheetData({ headers: [], rows: [], rowCount: 0 })
|
||||
setSpreadsheetData({ headers: [], rows: [], rowCount: 0, columnTypeMap: {} })
|
||||
setUploadedFile(null)
|
||||
}
|
||||
|
||||
@@ -102,9 +103,9 @@ const SpreadsheetImport: FC<Props> = ({
|
||||
message: `Multiple errors have been detected on ${errors.length} rows. Do check your input for any discrepancies.`,
|
||||
})
|
||||
}
|
||||
setSpreadsheetData({ headers, rows, rowCount: rows.length })
|
||||
setSpreadsheetData({ headers, rows, rowCount: rows.length, columnTypeMap: {} })
|
||||
} else {
|
||||
setSpreadsheetData({ headers: [], rows: [], rowCount: 0 })
|
||||
setSpreadsheetData({ headers: [], rows: [], rowCount: 0, columnTypeMap: {} })
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
+68
-5
@@ -1,4 +1,7 @@
|
||||
import dayjs from 'dayjs'
|
||||
import Papa from 'papaparse'
|
||||
import { has, includes } from 'lodash'
|
||||
import { tryParseJson } from 'lib/helpers'
|
||||
|
||||
const CHUNK_SIZE = 1024 * 1024 * 0.25 // 0.25MB
|
||||
|
||||
@@ -30,8 +33,10 @@ export const parseSpreadsheet = (file: File, onProgressUpdate: (progress: number
|
||||
let headers: string[] = []
|
||||
let chunkNumber = 0
|
||||
let rowCount = 0
|
||||
const rowPreview: any[] = []
|
||||
|
||||
const columnTypeMap: any = {}
|
||||
const errors: any[] = []
|
||||
|
||||
return new Promise((resolve) => {
|
||||
Papa.parse(file, {
|
||||
header: true,
|
||||
@@ -43,9 +48,14 @@ export const parseSpreadsheet = (file: File, onProgressUpdate: (progress: number
|
||||
chunk: (results) => {
|
||||
headers = results.meta.fields as string[]
|
||||
|
||||
if (rowCount === 0) {
|
||||
rowPreview.push(...results.data.slice(0, 50))
|
||||
}
|
||||
headers.forEach((header) => {
|
||||
const type = inferColumnType(header, results.data as any[])
|
||||
if (!has(columnTypeMap, header)) {
|
||||
columnTypeMap[header] = type
|
||||
} else if (columnTypeMap[header] !== type) {
|
||||
columnTypeMap[header] = 'text'
|
||||
}
|
||||
})
|
||||
|
||||
rowCount += results.data.length
|
||||
|
||||
@@ -58,7 +68,7 @@ export const parseSpreadsheet = (file: File, onProgressUpdate: (progress: number
|
||||
onProgressUpdate(progress > 1 ? 100 : Number((progress * 100).toFixed(2)))
|
||||
},
|
||||
complete: () => {
|
||||
const data = { headers, rowCount, rowPreview, errors }
|
||||
const data = { headers, rowCount, columnTypeMap, errors }
|
||||
resolve(data)
|
||||
},
|
||||
})
|
||||
@@ -68,3 +78,56 @@ export const parseSpreadsheet = (file: File, onProgressUpdate: (progress: number
|
||||
export const revertSpreadsheet = (headers: string[], rows: any[]) => {
|
||||
return Papa.unparse(rows, { columns: headers })
|
||||
}
|
||||
|
||||
const inferColumnType = (column: string, rows: object[]) => {
|
||||
// General strategy is to check the first row first, before checking across all the rows
|
||||
// to ensure uniformity in data type. Thinking we do this as an optimization instead of
|
||||
// checking all the rows up front.
|
||||
|
||||
// If there are no rows to infer for, default to text
|
||||
if (rows.length === 0) return 'text'
|
||||
|
||||
const columnData = (rows[0] as any)[column]
|
||||
const columnDataAcrossRows = rows.map((row: object) => (row as any)[column])
|
||||
|
||||
// Unable to infer any type as there's no data, default to text
|
||||
if (!columnData) {
|
||||
return 'text'
|
||||
}
|
||||
|
||||
// Infer numerical data type (defaults to either int8 or float8)
|
||||
if (Number(columnData)) {
|
||||
const columnNumberCheck = rows.map((row: object) => Number((row as any)[column]))
|
||||
if (columnNumberCheck.includes(NaN)) {
|
||||
return 'text'
|
||||
} else {
|
||||
const columnFloatCheck = columnNumberCheck.map((num: number) => num % 1)
|
||||
return columnFloatCheck.every((item) => item === 0) ? 'int8' : 'float8'
|
||||
}
|
||||
}
|
||||
|
||||
// Infer boolean type
|
||||
if (includes(['true', 'false'], columnData.toLowerCase())) {
|
||||
const isAllBoolean = columnDataAcrossRows.every((item: any) =>
|
||||
includes(['true', 'false'], item.toLowerCase())
|
||||
)
|
||||
if (isAllBoolean) {
|
||||
return 'boolean'
|
||||
}
|
||||
}
|
||||
|
||||
// Infer json type
|
||||
if (tryParseJson(columnData)) {
|
||||
const isAllJson = columnDataAcrossRows.every((item: any) => tryParseJson(columnData))
|
||||
if (isAllJson) {
|
||||
return 'jsonb'
|
||||
}
|
||||
}
|
||||
|
||||
// Infer datetime type
|
||||
if (dayjs(columnData, 'YYYY-MM-DD hh:mm:ss').isValid() && Date.parse(columnData)) {
|
||||
return 'timestamptz'
|
||||
}
|
||||
|
||||
return 'text'
|
||||
}
|
||||
+2
@@ -1,3 +1,4 @@
|
||||
import { Dictionary } from '@supabase/grid'
|
||||
import { ColumnField } from '../SidePanelEditor.types'
|
||||
|
||||
export interface TableField {
|
||||
@@ -13,4 +14,5 @@ export interface ImportContent {
|
||||
headers: string[]
|
||||
rowCount: number
|
||||
rows: object[]
|
||||
columnTypeMap: Dictionary<any>
|
||||
}
|
||||
+3
-57
@@ -1,8 +1,6 @@
|
||||
import { some, includes } from 'lodash'
|
||||
import dayjs from 'dayjs'
|
||||
import { some } from 'lodash'
|
||||
import { PostgresColumn, PostgresTable } from '@supabase/postgres-meta'
|
||||
|
||||
import { tryParseJson } from 'lib/helpers'
|
||||
import { ImportContent, TableField } from './TableEditor.types'
|
||||
import { DEFAULT_COLUMNS } from './TableEditor.constants'
|
||||
import { ColumnField } from '../SidePanelEditor.types'
|
||||
@@ -43,6 +41,7 @@ export const generateTableFieldFromPostgresTable = (
|
||||
id: table.id,
|
||||
name: isDuplicating ? `${table.name}_duplicate` : table.name,
|
||||
comment: isDuplicating ? `This is a duplicate of ${table.name}` : table?.comment ?? '',
|
||||
// @ts-ignore
|
||||
columns: table.columns.map((column: PostgresColumn) => {
|
||||
return generateColumnFieldFromPostgresColumn(column, table)
|
||||
}),
|
||||
@@ -52,61 +51,8 @@ export const generateTableFieldFromPostgresTable = (
|
||||
|
||||
export const formatImportedContentToColumnFields = (importContent: ImportContent) => {
|
||||
const columnFields = importContent.headers.map((header: string) => {
|
||||
const columnType = inferColumnType(header, importContent.rows)
|
||||
const columnType = importContent.columnTypeMap[header]
|
||||
return generateColumnField({ name: header, format: columnType })
|
||||
})
|
||||
return columnFields
|
||||
}
|
||||
|
||||
export const inferColumnType = (column: string, rows: object[]) => {
|
||||
// General strategy is to check the first row first, before checking across all the rows
|
||||
// to ensure uniformity in data type. Thinking we do this as an optimization instead of
|
||||
// checking all the rows up front.
|
||||
|
||||
// If there are no rows to infer for, default to text
|
||||
if (rows.length === 0) return 'text'
|
||||
|
||||
const columnData = (rows[0] as any)[column]
|
||||
const columnDataAcrossRows = rows.map((row: object) => (row as any)[column])
|
||||
|
||||
// Unable to infer any type as there's no data, default to text
|
||||
if (!columnData) {
|
||||
return 'text'
|
||||
}
|
||||
|
||||
// Infer numerical data type (defaults to either int8 or float8)
|
||||
if (Number(columnData)) {
|
||||
const columnNumberCheck = rows.map((row: object) => Number((row as any)[column]))
|
||||
if (columnNumberCheck.includes(NaN)) {
|
||||
return 'text'
|
||||
} else {
|
||||
const columnFloatCheck = columnNumberCheck.map((num: number) => num % 1)
|
||||
return columnFloatCheck.every((item) => item === 0) ? 'int8' : 'float8'
|
||||
}
|
||||
}
|
||||
|
||||
// Infer boolean type
|
||||
if (includes(['true', 'false'], columnData.toLowerCase())) {
|
||||
const isAllBoolean = columnDataAcrossRows.every((item: any) =>
|
||||
includes(['true', 'false'], item.toLowerCase())
|
||||
)
|
||||
if (isAllBoolean) {
|
||||
return 'boolean'
|
||||
}
|
||||
}
|
||||
|
||||
// Infer json type
|
||||
if (tryParseJson(columnData)) {
|
||||
const isAllJson = columnDataAcrossRows.every((item: any) => tryParseJson(columnData))
|
||||
if (isAllJson) {
|
||||
return 'jsonb'
|
||||
}
|
||||
}
|
||||
|
||||
// Infer datetime type
|
||||
if (dayjs(columnData, 'YYYY-MM-DD hh:mm:ss').isValid() && Date.parse(columnData)) {
|
||||
return 'timestamptz'
|
||||
}
|
||||
|
||||
return 'text'
|
||||
}
|
||||
@@ -75,7 +75,9 @@ const TableEditorMenu: FC<Props> = ({
|
||||
}
|
||||
|
||||
const schemas: PostgresSchema[] = meta.schemas.list()
|
||||
const tables: PostgresTable[] = meta.tables.list((table: PostgresTable) => table.schema === selectedSchema)
|
||||
const tables: PostgresTable[] = meta.tables.list(
|
||||
(table: PostgresTable) => table.schema === selectedSchema
|
||||
)
|
||||
|
||||
const schemaTables =
|
||||
searchText.length === 0
|
||||
@@ -89,7 +91,7 @@ const TableEditorMenu: FC<Props> = ({
|
||||
|
||||
// Temp fix - Ideally we'd just take up all the remaining space but
|
||||
// can't seem to figure that out immediately
|
||||
const maxScrollHeight = schemaViews.length > 0 ? 270 : 520
|
||||
const maxScrollHeight = schemaViews.length > 0 ? 270 : 515
|
||||
|
||||
return (
|
||||
<div className="my-6 flex flex-col flex-grow">
|
||||
|
||||
Reference in new issue
Block a user