Compare commits
139 Commits
dev
...
3.1.1-rele
| Author | SHA1 | Date |
|---|---|---|
|
|
26c30c9c34 | |
|
|
3293cfe162 | |
|
|
1c49d4afe3 | |
|
|
48ad9ff412 | |
|
|
efc0367999 | |
|
|
fe3e93861f | |
|
|
145c5ac8cb | |
|
|
525a6846ac | |
|
|
0fcbc5e401 | |
|
|
251756f410 | |
|
|
0156a64000 | |
|
|
b446b260f4 | |
|
|
9126e1d1fc | |
|
|
a8113f05ee | |
|
|
c7f990791e | |
|
|
4fcdba2c0c | |
|
|
cb3a38332f | |
|
|
7f3193643a | |
|
|
215c571c8d | |
|
|
e8edebc9d9 | |
|
|
43650bf721 | |
|
|
2481e68a19 | |
|
|
9957df1c41 | |
|
|
135d70da09 | |
|
|
0c3ab4335e | |
|
|
64360e5b69 | |
|
|
5eaf105290 | |
|
|
2c93dc6041 | |
|
|
94ef02e9de | |
|
|
cff3df1c45 | |
|
|
7e5a983f00 | |
|
|
c3a7af2d6e | |
|
|
eb449cdd44 | |
|
|
75f6c416fb | |
|
|
af031ef9f8 | |
|
|
826ca39a31 | |
|
|
5598ab01e9 | |
|
|
a5a849f615 | |
|
|
6309b78d37 | |
|
|
1e62855fa9 | |
|
|
eb7b48596c | |
|
|
d742994e62 | |
|
|
a17d0cc5d2 | |
|
|
7ca4438682 | |
|
|
58ef3cccd7 | |
|
|
541105a358 | |
|
|
36ba5a39c5 | |
|
|
a30f2ae8df | |
|
|
f3277277f0 | |
|
|
ab675dcf4b | |
|
|
344293102d | |
|
|
149553c52b | |
|
|
88094f914c | |
|
|
eeb11eedfb | |
|
|
a92f766843 | |
|
|
e04750b81d | |
|
|
dbd747951d | |
|
|
2b32c7b441 | |
|
|
3a14fd07b3 | |
|
|
35608becac | |
|
|
c79115c8d8 | |
|
|
2402b8a6ef | |
|
|
d2e56af838 | |
|
|
2a2802c938 | |
|
|
f0ad67f992 | |
|
|
703f9991b4 | |
|
|
5cfe6a96b4 | |
|
|
39c1144fab | |
|
|
b4f59af9d9 | |
|
|
b559d033d3 | |
|
|
10b7d24c2e | |
|
|
056176afc0 | |
|
|
771dd67b2e | |
|
|
7e39396a76 | |
|
|
ae33ba5947 | |
|
|
beebc5e0ad | |
|
|
f982e1d2a2 | |
|
|
3efe4bc308 | |
|
|
399f62cba2 | |
|
|
7a85b930d7 | |
|
|
a5a941e079 | |
|
|
3e899bee06 | |
|
|
42d8308940 | |
|
|
b0b29ed8e1 | |
|
|
71c51c3c3d | |
|
|
f871a4a41f | |
|
|
c47e088b73 | |
|
|
1e59250055 | |
|
|
1a63f8672a | |
|
|
e7b12bf205 | |
|
|
0d16d7b323 | |
|
|
a2b3659fe9 | |
|
|
f0fda2a9aa | |
|
|
280b7c8545 | |
|
|
cda3110409 | |
|
|
c286c5567a | |
|
|
f4babb773e | |
|
|
806ad1d98c | |
|
|
23d77f8a2a | |
|
|
34950d3c09 | |
|
|
780a509f67 | |
|
|
1513363eed | |
|
|
704043f229 | |
|
|
d7f40b19b5 | |
|
|
775ef98b64 | |
|
|
6f7ba2c634 | |
|
|
714e258be6 | |
|
|
8c6658e3f9 | |
|
|
2a437607ae | |
|
|
7ab4412b5e | |
|
|
9ad6a049c5 | |
|
|
7b58737e22 | |
|
|
0f3b42925f | |
|
|
89b4192d3d | |
|
|
216ceea641 | |
|
|
c11ec05d61 | |
|
|
fcc75ef1c6 | |
|
|
1aba077fb8 | |
|
|
52b79b017e | |
|
|
ed33066178 | |
|
|
f1de1707d5 | |
|
|
113dc4b5c5 | |
|
|
dde6f63c31 | |
|
|
cb063732d7 | |
|
|
f66dedc9da | |
|
|
1ac2e4a8f3 | |
|
|
6f3b4c1624 | |
|
|
e462918ac9 | |
|
|
1ea2e848cd | |
|
|
8eadf5e5aa | |
|
|
5e42f52bdf | |
|
|
c4953e8660 | |
|
|
951f707b61 | |
|
|
cd8f32d876 | |
|
|
f034a09d25 | |
|
|
0647b3e10c | |
|
|
27b69e608a | |
|
|
0f6cc4fe33 | |
|
|
8d1c2d3eeb |
|
|
@ -26,14 +26,12 @@
|
|||
/dolphinscheduler-dao/src/main/resources/sql/ @zhongjiajie
|
||||
/dolphinscheduler-common/ @caishunfeng
|
||||
/dolphinscheduler-standalone-server/ @kezhenxu94 @caishunfeng
|
||||
/dolphinscheduler-log-server/ @caishunfeng
|
||||
/dolphinscheduler-datasource-plugin/ @caishunfeng
|
||||
/dolphinscheduler-dist/ @kezhenxu94 @caishunfeng
|
||||
/dolphinscheduler-meter/ @caishunfeng @kezhenxu94 @ruanwenjun @EricGao888
|
||||
/dolphinscheduler-scheduler-plugin/ @caishunfeng
|
||||
/dolphinscheduler-master/ @caishunfeng @SbloodyS @ruanwenjun
|
||||
/dolphinscheduler-worker/ @caishunfeng @SbloodyS @ruanwenjun
|
||||
/dolphinscheduler-server/ @caishunfeng
|
||||
/dolphinscheduler-service/ @caishunfeng
|
||||
/dolphinscheduler-remote/ @caishunfeng
|
||||
/dolphinscheduler-spi/ @caishunfeng
|
||||
|
|
|
|||
|
|
@ -1,6 +1,5 @@
|
|||
<!--Thanks very much for contributing to Apache DolphinScheduler. Please review https://dolphinscheduler.apache.org/en-us/community/development/pull-request.html before opening a pull request.-->
|
||||
|
||||
|
||||
## Purpose of the pull request
|
||||
|
||||
<!--(For example: This pull request adds checkstyle plugin).-->
|
||||
|
|
@ -10,6 +9,7 @@
|
|||
<!--*(for example:)*
|
||||
- *Add maven-checkstyle-plugin to root pom.xml*
|
||||
-->
|
||||
|
||||
## Verify this pull request
|
||||
|
||||
<!--*(Please pick either of the following options)*-->
|
||||
|
|
|
|||
|
|
@ -26,12 +26,10 @@ backend:
|
|||
- 'dolphinscheduler-data-quality/**/*'
|
||||
- 'dolphinscheduler-datasource-plugin/**/*'
|
||||
- 'dolphinscheduler-dist/**/*'
|
||||
- 'dolphinscheduler-log-server/**/*'
|
||||
- 'dolphinscheduler-master/**/*'
|
||||
- 'dolphinscheduler-registry/**/*'
|
||||
- 'dolphinscheduler-remote/**/*'
|
||||
- 'dolphinscheduler-scheduler-plugin/**/*'
|
||||
- 'dolphinscheduler-server/**/*'
|
||||
- 'dolphinscheduler-service/**/*'
|
||||
- 'dolphinscheduler-spi/**/*'
|
||||
- 'dolphinscheduler-standalone-server/**/*'
|
||||
|
|
|
|||
|
|
@ -30,7 +30,6 @@ on:
|
|||
- 'dolphinscheduler-common/**'
|
||||
- 'dolphinscheduler-dao/**'
|
||||
- 'dolphinscheduler-rpc/**'
|
||||
- 'dolphinscheduler-server/**'
|
||||
pull_request:
|
||||
|
||||
concurrency:
|
||||
|
|
|
|||
|
|
@ -15,10 +15,10 @@
|
|||
# limitations under the License.
|
||||
#
|
||||
|
||||
FROM openjdk:8-jre-slim-buster
|
||||
FROM eclipse-temurin:8-jre
|
||||
|
||||
RUN apt update ; \
|
||||
apt install -y curl wget default-mysql-client sudo openssh-server netcat-traditional ;
|
||||
apt install -y wget default-mysql-client sudo openssh-server netcat-traditional ;
|
||||
|
||||
COPY ./apache-dolphinscheduler-*-SNAPSHOT-bin.tar.gz /root
|
||||
RUN tar -zxvf /root/apache-dolphinscheduler-*-SNAPSHOT-bin.tar.gz -C ~
|
||||
|
|
|
|||
|
|
@ -16,7 +16,7 @@
|
|||
#
|
||||
|
||||
# JAVA_HOME, will use it to start DolphinScheduler server
|
||||
export JAVA_HOME=${JAVA_HOME:-/usr/local/openjdk-8}
|
||||
export JAVA_HOME=${JAVA_HOME:-/opt/java/openjdk}
|
||||
|
||||
# Database related configuration, set database type, username and password
|
||||
export DATABASE=${DATABASE:-mysql}
|
||||
|
|
|
|||
|
|
@ -15,10 +15,10 @@
|
|||
# limitations under the License.
|
||||
#
|
||||
|
||||
FROM openjdk:8-jre-slim-buster
|
||||
FROM eclipse-temurin:8-jre
|
||||
|
||||
RUN apt update ; \
|
||||
apt install -y curl wget sudo openssh-server netcat-traditional ;
|
||||
apt install -y wget sudo openssh-server netcat-traditional ;
|
||||
|
||||
COPY ./apache-dolphinscheduler-*-SNAPSHOT-bin.tar.gz /root
|
||||
RUN tar -zxvf /root/apache-dolphinscheduler-*-SNAPSHOT-bin.tar.gz -C ~
|
||||
|
|
|
|||
|
|
@ -16,7 +16,7 @@
|
|||
#
|
||||
|
||||
# JAVA_HOME, will use it to start DolphinScheduler server
|
||||
export JAVA_HOME=${JAVA_HOME:-/usr/local/openjdk-8}
|
||||
export JAVA_HOME=${JAVA_HOME:-/opt/java/openjdk}
|
||||
|
||||
# Database related configuration, set database type, username and password
|
||||
export DATABASE=${DATABASE:-postgresql}
|
||||
|
|
|
|||
|
|
@ -43,7 +43,6 @@ dolphinscheduler-dao/src/main/resources/dao/data_source.properties
|
|||
dolphinscheduler-alert/logs/
|
||||
dolphinscheduler-alert/src/main/resources/alert.properties_bak
|
||||
dolphinscheduler-alert/src/main/resources/logback.xml
|
||||
dolphinscheduler-server/src/main/resources/logback.xml
|
||||
dolphinscheduler-ui/dist
|
||||
dolphinscheduler-ui/node
|
||||
dolphinscheduler-common/sql
|
||||
|
|
|
|||
|
|
@ -40,7 +40,6 @@ There will be two repositories at this time: origin (your own warehouse) and ups
|
|||
|
||||
Get/update remote repository code (already the latest code, skip it).
|
||||
|
||||
|
||||
```sh
|
||||
git fetch upstream
|
||||
```
|
||||
|
|
@ -91,7 +90,6 @@ After submitting changes to your remote repository, you should click on the new
|
|||
<img src = "http://geek.analysys.cn/static/upload/221/2019-04-02/90f3abbf-70ef-4334-b8d6-9014c9cf4c7f.png" width ="60%"/>
|
||||
</p>
|
||||
|
||||
|
||||
Select the modified local branch and the branch to merge past to create a pull request.
|
||||
|
||||
<p align = "center">
|
||||
|
|
|
|||
15
README.md
15
README.md
|
|
@ -1,6 +1,6 @@
|
|||
Dolphin Scheduler Official Website
|
||||
[dolphinscheduler.apache.org](https://dolphinscheduler.apache.org)
|
||||
============
|
||||
==================================================================
|
||||
|
||||
[](https://www.apache.org/licenses/LICENSE-2.0.html)
|
||||
[](https://codecov.io/gh/apache/dolphinscheduler/branch/dev)
|
||||
|
|
@ -8,9 +8,6 @@ Dolphin Scheduler Official Website
|
|||
[](https://twitter.com/dolphinschedule)
|
||||
[](https://s.apache.org/dolphinscheduler-slack)
|
||||
|
||||
|
||||
|
||||
|
||||
[](https://starchart.cc/apache/dolphinscheduler)
|
||||
|
||||
[](README.md)
|
||||
|
|
@ -45,11 +42,11 @@ scale of the cluster
|
|||
|
||||
## What's in DolphinScheduler
|
||||
|
||||
Stability | Accessibility | Features | Scalability |
|
||||
--------- | ------------- | -------- | ------------|
|
||||
Decentralized multi-master and multi-worker | Visualization of workflow key information, such as task status, task type, retry times, task operation machine information, visual variables, and so on at a glance. | Support pause, recover operation | Support customized task types
|
||||
support HA | Visualization of all workflow operations, dragging tasks to draw DAGs, configuring data sources and resources. At the same time, for third-party systems, provide API mode operations. | Users on DolphinScheduler can achieve many-to-one or one-to-one mapping relationship through tenants and Hadoop users, which is very important for scheduling large data jobs. | The scheduler supports distributed scheduling, and the overall scheduling capability will increase linearly with the scale of the cluster. Master and Worker support dynamic adjustment.
|
||||
Overload processing: By using the task queue mechanism, the number of schedulable tasks on a single machine can be flexibly configured. Machine jam can be avoided with high tolerance to numbers of tasks cached in task queue. | One-click deployment | Support traditional shell tasks, and big data platform task scheduling: MR, Spark, SQL (MySQL, PostgreSQL, hive, spark SQL), Python, Procedure, Sub_Process | |
|
||||
| Stability | Accessibility | Features | Scalability |
|
||||
|----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|--------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|
|
||||
| Decentralized multi-master and multi-worker | Visualization of workflow key information, such as task status, task type, retry times, task operation machine information, visual variables, and so on at a glance. | Support pause, recover operation | Support customized task types |
|
||||
| support HA | Visualization of all workflow operations, dragging tasks to draw DAGs, configuring data sources and resources. At the same time, for third-party systems, provide API mode operations. | Users on DolphinScheduler can achieve many-to-one or one-to-one mapping relationship through tenants and Hadoop users, which is very important for scheduling large data jobs. | The scheduler supports distributed scheduling, and the overall scheduling capability will increase linearly with the scale of the cluster. Master and Worker support dynamic adjustment. |
|
||||
| Overload processing: By using the task queue mechanism, the number of schedulable tasks on a single machine can be flexibly configured. Machine jam can be avoided with high tolerance to numbers of tasks cached in task queue. | One-click deployment | Support traditional shell tasks, and big data platform task scheduling: MR, Spark, SQL (MySQL, PostgreSQL, hive, spark SQL), Python, Procedure, Sub_Process | |
|
||||
|
||||
## User Interface Screenshots
|
||||
|
||||
|
|
|
|||
|
|
@ -1,12 +1,11 @@
|
|||
Dolphin Scheduler Official Website
|
||||
[dolphinscheduler.apache.org](https://dolphinscheduler.apache.org)
|
||||
============
|
||||
==================================================================
|
||||
|
||||
[](https://www.apache.org/licenses/LICENSE-2.0.html)
|
||||
[](https://codecov.io/gh/apache/dolphinscheduler/branch/dev)
|
||||
[](https://sonarcloud.io/dashboard?id=apache-dolphinscheduler)
|
||||
|
||||
|
||||
[](https://starchart.cc/apache/dolphinscheduler)
|
||||
|
||||
[](README_zh_CN.md)
|
||||
|
|
|
|||
|
|
@ -2,3 +2,4 @@
|
|||
|
||||
* [Start Up DolphinScheduler with Docker](https://dolphinscheduler.apache.org/en-us/docs/latest/user_doc/guide/start/docker.html)
|
||||
* [Start Up DolphinScheduler with Kubernetes](https://dolphinscheduler.apache.org/en-us/docs/latest/user_doc/guide/installation/kubernetes.html)
|
||||
|
||||
|
|
|
|||
|
|
@ -15,8 +15,8 @@
|
|||
# specific language governing permissions and limitations
|
||||
# under the License.
|
||||
#
|
||||
HUB=ghcr.io/apache/dolphinscheduler
|
||||
TAG=latest
|
||||
HUB=apache
|
||||
TAG=3.1.1
|
||||
|
||||
TZ=Asia/Shanghai
|
||||
DATABASE=postgresql
|
||||
|
|
|
|||
|
|
@ -35,11 +35,11 @@ type: application
|
|||
|
||||
# This is the chart version. This version number should be incremented each time you make changes
|
||||
# to the chart and its templates, including the app version.
|
||||
version: 2.0.0
|
||||
version: 3.1.1
|
||||
|
||||
# This is the version number of the application being deployed. This version number should be
|
||||
# incremented each time you make changes to the application.
|
||||
appVersion: dev-SNAPSHOT
|
||||
appVersion: 3.1.1
|
||||
|
||||
dependencies:
|
||||
- name: postgresql
|
||||
|
|
@ -56,3 +56,7 @@ dependencies:
|
|||
# Same as above.
|
||||
repository: https://raw.githubusercontent.com/bitnami/charts/archive-full-index/bitnami
|
||||
condition: zookeeper.enabled
|
||||
- name: mysql
|
||||
version: 9.4.1
|
||||
repository: https://raw.githubusercontent.com/bitnami/charts/archive-full-index/bitnami
|
||||
condition: mysql.enabled
|
||||
|
|
|
|||
|
|
@ -30,19 +30,19 @@ If release name contains chart name it will be used as a full name.
|
|||
Create default docker images' fullname.
|
||||
*/}}
|
||||
{{- define "dolphinscheduler.image.fullname.master" -}}
|
||||
{{- .Values.image.registry }}/dolphinscheduler-master:{{ .Values.image.tag | default .Chart.AppVersion -}}
|
||||
{{- .Values.image.registry }}/{{ .Values.image.master }}:{{ .Values.image.tag | default .Chart.AppVersion -}}
|
||||
{{- end -}}
|
||||
{{- define "dolphinscheduler.image.fullname.worker" -}}
|
||||
{{- .Values.image.registry }}/dolphinscheduler-worker:{{ .Values.image.tag | default .Chart.AppVersion -}}
|
||||
{{- .Values.image.registry }}/{{ .Values.image.worker }}:{{ .Values.image.tag | default .Chart.AppVersion -}}
|
||||
{{- end -}}
|
||||
{{- define "dolphinscheduler.image.fullname.api" -}}
|
||||
{{- .Values.image.registry }}/dolphinscheduler-api:{{ .Values.image.tag | default .Chart.AppVersion -}}
|
||||
{{- .Values.image.registry }}/{{ .Values.image.api }}:{{ .Values.image.tag | default .Chart.AppVersion -}}
|
||||
{{- end -}}
|
||||
{{- define "dolphinscheduler.image.fullname.alert" -}}
|
||||
{{- .Values.image.registry }}/dolphinscheduler-alert-server:{{ .Values.image.tag | default .Chart.AppVersion -}}
|
||||
{{- .Values.image.registry }}/{{ .Values.image.alert }}:{{ .Values.image.tag | default .Chart.AppVersion -}}
|
||||
{{- end -}}
|
||||
{{- define "dolphinscheduler.image.fullname.tools" -}}
|
||||
{{- .Values.image.registry }}/dolphinscheduler-tools:{{ .Values.image.tag | default .Chart.AppVersion -}}
|
||||
{{- .Values.image.registry }}/{{ .Values.image.tools }}:{{ .Values.image.tag | default .Chart.AppVersion -}}
|
||||
{{- end -}}
|
||||
|
||||
{{/*
|
||||
|
|
@ -100,7 +100,16 @@ We truncate at 63 chars because some Kubernetes name fields are limited to this
|
|||
{{- end -}}
|
||||
|
||||
{{/*
|
||||
Create a default fully qualified zookkeeper name.
|
||||
Create a default fully qualified mysql name.
|
||||
We truncate at 63 chars because some Kubernetes name fields are limited to this (by the DNS naming spec).
|
||||
*/}}
|
||||
{{- define "dolphinscheduler.mysql.fullname" -}}
|
||||
{{- $name := default "mysql" .Values.mysql.nameOverride -}}
|
||||
{{- printf "%s-%s" .Release.Name $name | trunc 63 | trimSuffix "-" -}}
|
||||
{{- end -}}
|
||||
|
||||
{{/*
|
||||
Create a default fully qualified zookeeper name.
|
||||
We truncate at 63 chars because some Kubernetes name fields are limited to this (by the DNS naming spec).
|
||||
*/}}
|
||||
{{- define "dolphinscheduler.zookeeper.fullname" -}}
|
||||
|
|
@ -123,18 +132,24 @@ Create a database environment variables.
|
|||
- name: DATABASE
|
||||
{{- if .Values.postgresql.enabled }}
|
||||
value: "postgresql"
|
||||
{{- else if .Values.mysql.enabled }}
|
||||
value: "mysql"
|
||||
{{- else }}
|
||||
value: {{ .Values.externalDatabase.type | quote }}
|
||||
{{- end }}
|
||||
- name: SPRING_DATASOURCE_URL
|
||||
{{- if .Values.postgresql.enabled }}
|
||||
value: jdbc:postgresql://{{ template "dolphinscheduler.postgresql.fullname" . }}:5432/{{ .Values.postgresql.postgresqlDatabase }}?characterEncoding=utf8
|
||||
{{- else if .Values.mysql.enabled }}
|
||||
value: jdbc:mysql://{{ template "dolphinscheduler.mysql.fullname" . }}:3306/{{ .Values.postgresql.postgresqlDatabase }}?characterEncoding=utf8
|
||||
{{- else }}
|
||||
value: jdbc:{{ .Values.externalDatabase.type }}://{{ .Values.externalDatabase.host }}:{{ .Values.externalDatabase.port }}/{{ .Values.externalDatabase.database }}?{{ .Values.externalDatabase.params }}
|
||||
{{- end }}
|
||||
- name: SPRING_DATASOURCE_USERNAME
|
||||
{{- if .Values.postgresql.enabled }}
|
||||
value: {{ .Values.postgresql.postgresqlUsername }}
|
||||
{{- else if .Values.mysql.enabled }}
|
||||
value: {{ .Values.mysql.auth.username }}
|
||||
{{- else }}
|
||||
value: {{ .Values.externalDatabase.username | quote }}
|
||||
{{- end }}
|
||||
|
|
@ -144,6 +159,9 @@ Create a database environment variables.
|
|||
{{- if .Values.postgresql.enabled }}
|
||||
name: {{ template "dolphinscheduler.postgresql.fullname" . }}
|
||||
key: postgresql-password
|
||||
{{- else if .Values.mysql.enabled }}
|
||||
name: {{ template "dolphinscheduler.mysql.fullname" . }}
|
||||
key: mysql-password
|
||||
{{- else }}
|
||||
name: {{ include "dolphinscheduler.fullname" . }}-externaldb
|
||||
key: database-password
|
||||
|
|
@ -159,6 +177,8 @@ Wait for database to be ready.
|
|||
imagePullPolicy: IfNotPresent
|
||||
{{- if .Values.postgresql.enabled }}
|
||||
command: ['sh', '-xc', 'for i in $(seq 1 180); do nc -z -w3 {{ template "dolphinscheduler.postgresql.fullname" . }} 5432 && exit 0 || sleep 5; done; exit 1']
|
||||
{{- else if .Values.mysql.enabled }}
|
||||
command: ['sh', '-xc', 'for i in $(seq 1 180); do nc -z -w3 {{ template "dolphinscheduler.mysql.fullname" . }} 3306 && exit 0 || sleep 5; done; exit 1']
|
||||
{{- else }}
|
||||
command: ['sh', '-xc', 'for i in $(seq 1 180); do nc -z -w3 {{ .Values.externalDatabase.host }} {{ .Values.externalDatabase.port }} && exit 0 || sleep 5; done; exit 1']
|
||||
{{- end }}
|
||||
|
|
@ -182,19 +202,6 @@ Create a registry environment variables.
|
|||
{{- end }}
|
||||
{{- end -}}
|
||||
|
||||
{{/*
|
||||
Create a common fs_s3a environment variables.
|
||||
*/}}
|
||||
{{- define "dolphinscheduler.fs_s3a.env_vars" -}}
|
||||
{{- if eq (default "HDFS" .Values.common.configmap.RESOURCE_STORAGE_TYPE) "S3" -}}
|
||||
- name: FS_S3A_SECRET_KEY
|
||||
valueFrom:
|
||||
secretKeyRef:
|
||||
key: fs-s3a-secret-key
|
||||
name: {{ include "dolphinscheduler.fullname" . }}-fs-s3a
|
||||
{{- end -}}
|
||||
{{- end -}}
|
||||
|
||||
{{/*
|
||||
Create a sharedStoragePersistence volume.
|
||||
*/}}
|
||||
|
|
|
|||
|
|
@ -23,7 +23,7 @@ metadata:
|
|||
app.kubernetes.io/name: {{ include "dolphinscheduler.fullname" . }}-common
|
||||
{{- include "dolphinscheduler.common.labels" . | nindent 4 }}
|
||||
data:
|
||||
{{- range $key, $value := (omit .Values.common.configmap "FS_S3A_SECRET_KEY") }}
|
||||
{{- range $key, $value := .Values.common.configmap }}
|
||||
{{ $key }}: {{ $value | quote }}
|
||||
{{- end }}
|
||||
{{- end }}
|
||||
|
|
@ -39,6 +39,7 @@ spec:
|
|||
{{- toYaml .Values.alert.annotations | nindent 8 }}
|
||||
{{- end }}
|
||||
spec:
|
||||
serviceAccountName: {{ template "dolphinscheduler.fullname" . }}
|
||||
{{- if .Values.alert.affinity }}
|
||||
affinity:
|
||||
{{- toYaml .Values.alert.affinity | nindent 8 }}
|
||||
|
|
@ -70,6 +71,7 @@ spec:
|
|||
- name: SPRING_JACKSON_TIME_ZONE
|
||||
value: {{ .Values.timezone }}
|
||||
{{- include "dolphinscheduler.database.env_vars" . | nindent 12 }}
|
||||
{{- include "dolphinscheduler.registry.env_vars" . | nindent 12 }}
|
||||
{{ range $key, $value := .Values.alert.env }}
|
||||
- name: {{ $key }}
|
||||
value: {{ $value | quote }}
|
||||
|
|
|
|||
|
|
@ -39,6 +39,7 @@ spec:
|
|||
{{- toYaml .Values.api.annotations | nindent 8 }}
|
||||
{{- end }}
|
||||
spec:
|
||||
serviceAccountName: {{ template "dolphinscheduler.fullname" . }}
|
||||
{{- if .Values.api.affinity }}
|
||||
affinity:
|
||||
{{- toYaml .Values.api.affinity | nindent 8 }}
|
||||
|
|
@ -71,7 +72,6 @@ spec:
|
|||
value: {{ .Values.timezone }}
|
||||
{{- include "dolphinscheduler.database.env_vars" . | nindent 12 }}
|
||||
{{- include "dolphinscheduler.registry.env_vars" . | nindent 12 }}
|
||||
{{- include "dolphinscheduler.fs_s3a.env_vars" . | nindent 12 }}
|
||||
{{ range $key, $value := .Values.api.env }}
|
||||
- name: {{ $key }}
|
||||
value: {{ $value | quote }}
|
||||
|
|
|
|||
|
|
@ -34,7 +34,7 @@ metadata:
|
|||
{{- end }}
|
||||
spec:
|
||||
rules:
|
||||
- host: {{ .Values.ingress.host }}
|
||||
- host: "{{ .Values.ingress.host }}"
|
||||
http:
|
||||
paths:
|
||||
- path: {{ .Values.ingress.path }}
|
||||
|
|
|
|||
|
|
@ -46,7 +46,6 @@ spec:
|
|||
value: {{ .Values.timezone }}
|
||||
{{- include "dolphinscheduler.database.env_vars" . | nindent 12 }}
|
||||
{{- include "dolphinscheduler.registry.env_vars" . | nindent 12 }}
|
||||
{{- include "dolphinscheduler.fs_s3a.env_vars" . | nindent 12 }}
|
||||
envFrom:
|
||||
- configMapRef:
|
||||
name: {{ include "dolphinscheduler.fullname" . }}-common
|
||||
|
|
|
|||
|
|
@ -0,0 +1,53 @@
|
|||
# Licensed to the Apache Software Foundation (ASF) under one or more
|
||||
# contributor license agreements. See the NOTICE file distributed with
|
||||
# this work for additional information regarding copyright ownership.
|
||||
# The ASF licenses this file to You under the Apache License, Version 2.0
|
||||
# (the "License"); you may not use this file except in compliance with
|
||||
# the License. You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
|
||||
apiVersion: v1
|
||||
kind: ServiceAccount
|
||||
metadata:
|
||||
labels:
|
||||
app: {{ template "dolphinscheduler.fullname" . }}
|
||||
chart: {{ .Chart.Name }}-{{ .Chart.Version }}
|
||||
release: {{ .Release.Name }}
|
||||
name: {{ template "dolphinscheduler.fullname" . }}
|
||||
---
|
||||
kind: Role
|
||||
apiVersion: rbac.authorization.k8s.io/v1
|
||||
metadata:
|
||||
name: {{ template "dolphinscheduler.fullname" . }}
|
||||
labels:
|
||||
app: {{ template "dolphinscheduler.fullname" . }}
|
||||
chart: "{{ .Chart.Name }}-{{ .Chart.Version }}"
|
||||
release: "{{ .Release.Name }}"
|
||||
rules:
|
||||
- apiGroups: [""]
|
||||
resources: ["configmaps"]
|
||||
verbs: ["get", "watch", "list"]
|
||||
---
|
||||
apiVersion: rbac.authorization.k8s.io/v1
|
||||
kind: RoleBinding
|
||||
metadata:
|
||||
name: {{ template "dolphinscheduler.fullname" . }}
|
||||
labels:
|
||||
app: {{ template "dolphinscheduler.fullname" . }}
|
||||
chart: "{{ .Chart.Name }}-{{ .Chart.Version }}"
|
||||
release: "{{ .Release.Name }}"
|
||||
roleRef:
|
||||
apiGroup: rbac.authorization.k8s.io
|
||||
kind: Role
|
||||
name: {{ template "dolphinscheduler.fullname" . }}
|
||||
subjects:
|
||||
- kind: ServiceAccount
|
||||
name: {{ template "dolphinscheduler.fullname" . }}
|
||||
namespace: {{ .Release.Namespace }}
|
||||
|
|
@ -1,28 +0,0 @@
|
|||
#
|
||||
# Licensed to the Apache Software Foundation (ASF) under one or more
|
||||
# contributor license agreements. See the NOTICE file distributed with
|
||||
# this work for additional information regarding copyright ownership.
|
||||
# The ASF licenses this file to You under the Apache License, Version 2.0
|
||||
# (the "License"); you may not use this file except in compliance with
|
||||
# the License. You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
#
|
||||
{{- if eq (default "HDFS" .Values.common.configmap.RESOURCE_STORAGE_TYPE) "S3" }}
|
||||
apiVersion: v1
|
||||
kind: Secret
|
||||
metadata:
|
||||
name: {{ include "dolphinscheduler.fullname" . }}-fs-s3a
|
||||
labels:
|
||||
app.kubernetes.io/name: {{ include "dolphinscheduler.fullname" . }}-fs-s3a
|
||||
{{- include "dolphinscheduler.common.labels" . | nindent 4 }}
|
||||
type: Opaque
|
||||
data:
|
||||
fs-s3a-secret-key: {{ .Values.common.configmap.FS_S3A_SECRET_KEY | b64enc | quote }}
|
||||
{{- end }}
|
||||
|
|
@ -36,6 +36,7 @@ spec:
|
|||
{{- toYaml .Values.master.annotations | nindent 8 }}
|
||||
{{- end }}
|
||||
spec:
|
||||
serviceAccountName: {{ template "dolphinscheduler.fullname" . }}
|
||||
{{- if .Values.master.affinity }}
|
||||
affinity:
|
||||
{{- toYaml .Values.master.affinity | nindent 8 }}
|
||||
|
|
@ -66,7 +67,6 @@ spec:
|
|||
value: {{ .Values.timezone }}
|
||||
{{- include "dolphinscheduler.database.env_vars" . | nindent 12 }}
|
||||
{{- include "dolphinscheduler.registry.env_vars" . | nindent 12 }}
|
||||
{{- include "dolphinscheduler.fs_s3a.env_vars" . | nindent 12 }}
|
||||
{{ range $key, $value := .Values.master.env }}
|
||||
- name: {{ $key }}
|
||||
value: {{ $value | quote }}
|
||||
|
|
|
|||
|
|
@ -36,6 +36,7 @@ spec:
|
|||
{{- toYaml .Values.worker.annotations | nindent 8 }}
|
||||
{{- end }}
|
||||
spec:
|
||||
serviceAccountName: {{ template "dolphinscheduler.fullname" . }}
|
||||
{{- if .Values.worker.affinity }}
|
||||
affinity:
|
||||
{{- toYaml .Values.worker.affinity | nindent 8 }}
|
||||
|
|
@ -68,7 +69,6 @@ spec:
|
|||
value: {{ include "dolphinscheduler.fullname" . }}-alert
|
||||
{{- include "dolphinscheduler.database.env_vars" . | nindent 12 }}
|
||||
{{- include "dolphinscheduler.registry.env_vars" . | nindent 12 }}
|
||||
{{- include "dolphinscheduler.fs_s3a.env_vars" . | nindent 12 }}
|
||||
{{ range $key, $value := .Values.worker.env }}
|
||||
- name: {{ $key }}
|
||||
value: {{ $value | quote }}
|
||||
|
|
|
|||
|
|
@ -42,8 +42,8 @@ spec:
|
|||
name: api-port
|
||||
- port: 25333
|
||||
targetPort: python-api-port
|
||||
{{- if and (eq .Values.api.service.type "NodePort") .Values.api.service.nodePort }}
|
||||
nodePort: {{ .Values.api.service.nodePort }}
|
||||
{{- if and (eq .Values.api.service.type "NodePort") .Values.api.service.pythonNodePort }}
|
||||
nodePort: {{ .Values.api.service.pythonNodePort }}
|
||||
{{- end }}
|
||||
protocol: TCP
|
||||
name: python-api-port
|
||||
|
|
|
|||
|
|
@ -23,9 +23,14 @@ timezone: "Asia/Shanghai"
|
|||
|
||||
image:
|
||||
registry: "dolphinscheduler.docker.scarf.sh/apache"
|
||||
tag: "dev-SNAPSHOT"
|
||||
tag: "3.1.1"
|
||||
pullPolicy: "IfNotPresent"
|
||||
pullSecret: ""
|
||||
master: dolphinscheduler-master
|
||||
worker: dolphinscheduler-worker
|
||||
api: dolphinscheduler-api
|
||||
alert: dolphinscheduler-alert-server
|
||||
tools: dolphinscheduler-tools
|
||||
|
||||
## If not exists external database, by default, Dolphinscheduler's database will use it.
|
||||
postgresql:
|
||||
|
|
@ -38,6 +43,18 @@ postgresql:
|
|||
size: "20Gi"
|
||||
storageClass: "-"
|
||||
|
||||
mysql:
|
||||
enabled: false
|
||||
auth:
|
||||
username: "ds"
|
||||
password: "ds"
|
||||
database: "dolphinscheduler"
|
||||
primary:
|
||||
persistence:
|
||||
enabled: false
|
||||
size: "20Gi"
|
||||
storageClass: "-"
|
||||
|
||||
## If exists external database, and set postgresql.enable value to false.
|
||||
## external database will be used, otherwise Dolphinscheduler's database will be used.
|
||||
externalDatabase:
|
||||
|
|
@ -74,8 +91,8 @@ conf:
|
|||
# resource storage type: HDFS, S3, NONE
|
||||
resource.storage.type: HDFS
|
||||
|
||||
# resource store on HDFS/S3 path, resource file will store to this hadoop hdfs path, self configuration, please make sure the directory exists on hdfs and have read write permissions. "/dolphinscheduler" is recommended
|
||||
resource.upload.path: /dolphinscheduler
|
||||
# resource store on HDFS/S3 path, resource file will store to this base path, self configuration, please make sure the directory exists on hdfs and have read write permissions. "/dolphinscheduler" is recommended
|
||||
resource.storage.upload.base.path: /dolphinscheduler
|
||||
|
||||
# whether to startup kerberos
|
||||
hadoop.security.authentication.startup.state: false
|
||||
|
|
@ -93,14 +110,20 @@ conf:
|
|||
kerberos.expire.time: 2
|
||||
# resource view suffixs
|
||||
#resource.view.suffixs: txt,log,sh,bat,conf,cfg,py,java,sql,xml,hql,properties,json,yml,yaml,ini,js
|
||||
# if resource.storage.type: HDFS, the user must have the permission to create directories under the HDFS root path
|
||||
hdfs.root.user: hdfs
|
||||
# if resource.storage.type: S3, the value like: s3a://dolphinscheduler; if resource.storage.type: HDFS and namenode HA is enabled, you need to copy core-site.xml and hdfs-site.xml to conf dir
|
||||
fs.defaultFS: file:///
|
||||
aws.access.key.id: minioadmin
|
||||
aws.secret.access.key: minioadmin
|
||||
aws.region: us-east-1
|
||||
aws.endpoint: http://localhost:9000
|
||||
# if resource.storage.type=HDFS, the user must have the permission to create directories under the HDFS root path
|
||||
resource.hdfs.root.user: hdfs
|
||||
# if resource.storage.type=S3, the value like: s3a://dolphinscheduler; if resource.storage.type=HDFS and namenode HA is enabled, you need to copy core-site.xml and hdfs-site.xml to conf dir
|
||||
resource.hdfs.fs.defaultFS: hdfs://mycluster:8020
|
||||
# The AWS access key. if resource.storage.type=S3 or use EMR-Task, This configuration is required
|
||||
resource.aws.access.key.id: minioadmin
|
||||
# The AWS secret access key. if resource.storage.type=S3 or use EMR-Task, This configuration is required
|
||||
resource.aws.secret.access.key: minioadmin
|
||||
# The AWS Region to use. if resource.storage.type=S3 or use EMR-Task, This configuration is required
|
||||
resource.aws.region: cn-north-1
|
||||
# The name of the bucket. You need to create them by yourself. Otherwise, the system cannot start. All buckets in Amazon S3 share a single namespace; ensure the bucket is given a unique name.
|
||||
resource.aws.s3.bucket.name: dolphinscheduler
|
||||
# You need to set this parameter when private cloud s3. If S3 uses public cloud, you only need to set resource.aws.region or set to the endpoint of a public cloud such as S3.cn-north-1.amazonaws.com.cn
|
||||
resource.aws.s3.endpoint: http://localhost:9000
|
||||
# resourcemanager port, the default value is 8088 if not specified
|
||||
resource.manager.httpaddress.port: 8088
|
||||
# if resourcemanager HA is enabled, please set the HA IPs; if resourcemanager is single, keep this value empty
|
||||
|
|
@ -149,25 +172,8 @@ common:
|
|||
configmap:
|
||||
DOLPHINSCHEDULER_OPTS: ""
|
||||
DATA_BASEDIR_PATH: "/tmp/dolphinscheduler"
|
||||
RESOURCE_STORAGE_TYPE: "HDFS"
|
||||
RESOURCE_UPLOAD_PATH: "/dolphinscheduler"
|
||||
FS_DEFAULT_FS: "file:///"
|
||||
FS_S3A_ENDPOINT: "s3.xxx.amazonaws.com"
|
||||
FS_S3A_ACCESS_KEY: "xxxxxxx"
|
||||
FS_S3A_SECRET_KEY: "xxxxxxx"
|
||||
HADOOP_SECURITY_AUTHENTICATION_STARTUP_STATE: "false"
|
||||
JAVA_SECURITY_KRB5_CONF_PATH: "/opt/krb5.conf"
|
||||
LOGIN_USER_KEYTAB_USERNAME: "hdfs@HADOOP.COM"
|
||||
LOGIN_USER_KEYTAB_PATH: "/opt/hdfs.keytab"
|
||||
KERBEROS_EXPIRE_TIME: "2"
|
||||
HDFS_ROOT_USER: "hdfs"
|
||||
RESOURCE_MANAGER_HTTPADDRESS_PORT: "8088"
|
||||
YARN_RESOURCEMANAGER_HA_RM_IDS: ""
|
||||
YARN_APPLICATION_STATUS_ADDRESS: "http://ds1:%s/ws/v1/cluster/apps/%s"
|
||||
YARN_JOB_HISTORY_STATUS_ADDRESS: "http://ds1:19888/ws/v1/history/mapreduce/jobs/%s"
|
||||
DATASOURCE_ENCRYPTION_ENABLE: "false"
|
||||
DATASOURCE_ENCRYPTION_SALT: "!@#$%^&*"
|
||||
SUDO_ENABLE: "true"
|
||||
|
||||
# dolphinscheduler env
|
||||
HADOOP_HOME: "/opt/soft/hadoop"
|
||||
HADOOP_CONF_DIR: "/opt/soft/hadoop/etc/hadoop"
|
||||
|
|
@ -468,8 +474,10 @@ api:
|
|||
type: "ClusterIP"
|
||||
## clusterIP is the IP address of the service and is usually assigned randomly by the master
|
||||
clusterIP: ""
|
||||
## nodePort is the port on each node on which this service is exposed when type=NodePort
|
||||
## nodePort is the port on each node on which this api service is exposed when type=NodePort
|
||||
nodePort: ""
|
||||
## pythonNodePort is the port on each node on which this python api service is exposed when type=NodePort
|
||||
pythonNodePort: ""
|
||||
## externalIPs is a list of IP addresses for which nodes in the cluster will also accept traffic for this service
|
||||
externalIPs: []
|
||||
## externalName is the external reference that kubedns or equivalent will return as a CNAME record for this service, requires Type to be ExternalName
|
||||
|
|
|
|||
File diff suppressed because it is too large
Load Diff
|
|
@ -45,11 +45,16 @@ export default {
|
|||
children: [
|
||||
{
|
||||
key: 'docs0',
|
||||
text: 'latest(3.0.0)',
|
||||
text: 'latest(3.1.1)',
|
||||
link: '/en-us/docs/latest/user_doc/about/introduction.html',
|
||||
},
|
||||
{
|
||||
key: 'docs1',
|
||||
text: '3.0.1',
|
||||
link: '/en-us/docs/3.0.1/user_doc/about/introduction.html',
|
||||
},
|
||||
{
|
||||
key: 'docs2',
|
||||
text: '2.0.6',
|
||||
link: '/en-us/docs/2.0.6/user_doc/guide/quick-start.html',
|
||||
},
|
||||
|
|
|
|||
|
|
@ -89,3 +89,4 @@ closed and transfer from [current DSIPs][current-DSIPs] to [past DSIPs][past-DSI
|
|||
[github-issue-choose]: https://github.com/apache/dolphinscheduler/issues/new/choose
|
||||
[mail-to-dev]: mailto:dev@dolphinscheduler.apache.org
|
||||
[DSIP-1]: https://github.com/apache/dolphinscheduler/issues/6407
|
||||
|
||||
|
|
|
|||
|
|
@ -17,3 +17,4 @@
|
|||
## High Scalability
|
||||
|
||||
- **Scalability**: Supports multitenancy and online resource management. Stable operation of 100,000 data tasks per day is supported.
|
||||
|
||||
|
|
|
|||
|
|
@ -17,10 +17,10 @@ there are no subsequent nodes. Examples are as follows:
|
|||
manual start or scheduled scheduling. Each time the process definition runs, a process instance is generated
|
||||
|
||||
**Task instance**: The task instance is the instantiation of the task node in the process definition, which identifies
|
||||
the specific task execution status
|
||||
the specific task
|
||||
|
||||
**Task type**: Currently supports SHELL, SQL, SUB_PROCESS (sub-process), PROCEDURE, MR, SPARK, PYTHON, DEPENDENT (
|
||||
depends), and plans to support dynamic plug-in expansion, note: **SUB_PROCESS** It is also a separate process
|
||||
depends), and plans to support dynamic plug-in expansion, note: **SUB_PROCESS** need relation with another workflow definition which also a separate process
|
||||
definition that can be started and executed separately
|
||||
|
||||
**Scheduling method**: The system supports scheduled scheduling and manual scheduling based on cron expressions. Command
|
||||
|
|
@ -45,10 +45,14 @@ provided. **Continue** refers to regardless of the status of the task running in
|
|||
failure. **End** means that once a failed task is found, Kill will also run the parallel task at the same time, and the
|
||||
process fails and ends
|
||||
|
||||
**Complement**: Supplement historical data,supports **interval parallel and serial** two complement methods, and two types of date selection which include **date range** and **date enumeration**.
|
||||
**Complement**: Supplement historical data,supports **interval parallel** and **serial** two complement methods, and two types of date selection which include **date range** and **date enumeration**.
|
||||
|
||||
### 2.Module introduction
|
||||
|
||||
- dolphinscheduler-master master module, provides workflow management and orchestration.
|
||||
|
||||
- dolphinscheduler-worker worker module, provides task execution management.
|
||||
|
||||
- dolphinscheduler-alert alarm module, providing AlertServer service.
|
||||
|
||||
- dolphinscheduler-api web application module, providing ApiServer service.
|
||||
|
|
@ -59,8 +63,6 @@ process fails and ends
|
|||
|
||||
- dolphinscheduler-remote client and server based on netty
|
||||
|
||||
- dolphinscheduler-server MasterServer and WorkerServer services
|
||||
|
||||
- dolphinscheduler-service service module, including Quartz, Zookeeper, log client access service, easy to call server
|
||||
module and api module
|
||||
|
||||
|
|
@ -71,4 +73,3 @@ process fails and ends
|
|||
From the perspective of scheduling, this article preliminarily introduces the architecture principles and implementation
|
||||
ideas of the big data distributed workflow scheduling system-DolphinScheduler. To be continued
|
||||
|
||||
|
||||
|
|
|
|||
|
|
@ -7,7 +7,7 @@ This section briefs about the hardware requirements for DolphinScheduler. Dolphi
|
|||
The Linux operating systems specified below can run on physical servers and mainstream virtualization environments such as VMware, KVM, and XEN.
|
||||
|
||||
| Operating System | Version |
|
||||
| :----------------------- | :----------: |
|
||||
|:-------------------------|:---------------:|
|
||||
| Red Hat Enterprise Linux | 7.0 and above |
|
||||
| CentOS | 7.0 and above |
|
||||
| Oracle Enterprise Linux | 7.0 and above |
|
||||
|
|
@ -23,7 +23,7 @@ DolphinScheduler supports 64-bit hardware platforms with Intel x86-64 architectu
|
|||
### Production Environment
|
||||
|
||||
| **CPU** | **MEM** | **HD** | **NIC** | **Num** |
|
||||
| --- | --- | --- | --- | --- |
|
||||
|---------|---------|--------|---------|---------|
|
||||
| 4 core+ | 8 GB+ | SAS | GbE | 1+ |
|
||||
|
||||
> **Note:**
|
||||
|
|
@ -35,7 +35,7 @@ DolphinScheduler supports 64-bit hardware platforms with Intel x86-64 architectu
|
|||
DolphinScheduler provides the following network port configurations for normal operation:
|
||||
|
||||
| Server | Port | Desc |
|
||||
| --- | --- | --- |
|
||||
|----------------------|-------|----------------------------------------------------------------------|
|
||||
| MasterServer | 5678 | not the communication port, require the native ports do not conflict |
|
||||
| WorkerServer | 1234 | not the communication port, require the native ports do not conflict |
|
||||
| ApiApplicationServer | 12345 | backend communication port |
|
||||
|
|
|
|||
|
|
@ -101,8 +101,6 @@ The directory structure of DolphinScheduler is as follows:
|
|||
|
||||
## Configurations in Details
|
||||
|
||||
|
||||
|
||||
### dolphinscheduler-daemon.sh [startup or shutdown DolphinScheduler application]
|
||||
|
||||
dolphinscheduler-daemon.sh is responsible for DolphinScheduler startup and shutdown.
|
||||
|
|
@ -110,6 +108,7 @@ Essentially, start-all.sh or stop-all.sh startup and shutdown the cluster via do
|
|||
Currently, DolphinScheduler just makes a basic config, remember to config further JVM options based on your practical situation of resources.
|
||||
|
||||
Default simplified parameters are:
|
||||
|
||||
```bash
|
||||
export DOLPHINSCHEDULER_OPTS="
|
||||
-server
|
||||
|
|
@ -157,8 +156,8 @@ The default configuration is as follows:
|
|||
|
||||
Note that DolphinScheduler also supports database configuration through `bin/env/dolphinscheduler_env.sh`.
|
||||
|
||||
|
||||
### Zookeeper related configuration
|
||||
|
||||
DolphinScheduler uses Zookeeper for cluster management, fault tolerance, event monitoring and other functions. Configuration file location:
|
||||
|Service| Configuration file |
|
||||
|--|--|
|
||||
|
|
@ -226,8 +225,8 @@ The default configuration is as follows:
|
|||
|alert.rpc.port | 50052 | the RPC port of Alert Server|
|
||||
|zeppelin.rest.url | http://localhost:8080 | the RESTful API url of zeppelin|
|
||||
|
||||
|
||||
### Api-server related configuration
|
||||
|
||||
Location: `api-server/conf/application.yaml`
|
||||
|
||||
|Parameters | Default value| Description|
|
||||
|
|
@ -257,6 +256,7 @@ Location: `api-server/conf/application.yaml`
|
|||
|traffic.control.customize-tenant-qps-rate||customize tenant max request number per second|
|
||||
|
||||
### Master Server related configuration
|
||||
|
||||
Location: `master-server/conf/application.yaml`
|
||||
|
||||
|Parameters | Default value| Description|
|
||||
|
|
@ -277,9 +277,10 @@ Location: `master-server/conf/application.yaml`
|
|||
|master.kill-yarn-job-when-task-failover|true|whether to kill yarn job when failover taskInstance|
|
||||
|master.registry-disconnect-strategy.strategy|stop|Used when the master disconnect from registry, default value: stop. Optional values include stop, waiting|
|
||||
|master.registry-disconnect-strategy.max-waiting-time|100s|Used when the master disconnect from registry, and the disconnect strategy is waiting, this config means the master will waiting to reconnect to registry in given times, and after the waiting times, if the master still cannot connect to registry, will stop itself, if the value is 0s, the Master will waitting infinitely|
|
||||
|
||||
|master.worker-group-refresh-interval|10s|The interval to refresh worker group from db to memory|
|
||||
|
||||
### Worker Server related configuration
|
||||
|
||||
Location: `worker-server/conf/application.yaml`
|
||||
|
||||
|Parameters | Default value| Description|
|
||||
|
|
@ -291,13 +292,14 @@ Location: `worker-server/conf/application.yaml`
|
|||
|worker.tenant-auto-create|true|tenant corresponds to the user of the system, which is used by the worker to submit the job. If system does not have this user, it will be automatically created after the parameter worker.tenant.auto.create is true.|
|
||||
|worker.max-cpu-load-avg|-1|worker max CPU load avg, only higher than the system CPU load average, worker server can be dispatched tasks. default value -1: the number of CPU cores * 2|
|
||||
|worker.reserved-memory|0.3|worker reserved memory, only lower than system available memory, worker server can be dispatched tasks. default value 0.3, the unit is G|
|
||||
|worker.groups|default|worker groups separated by comma, e.g., 'worker.groups=default,test' <br> worker will join corresponding group according to this config when startup|
|
||||
|worker.alert-listen-host|localhost|the alert listen host of worker|
|
||||
|worker.alert-listen-port|50052|the alert listen port of worker|
|
||||
|worker.registry-disconnect-strategy.strategy|stop|Used when the worker disconnect from registry, default value: stop. Optional values include stop, waiting|
|
||||
|worker.registry-disconnect-strategy.max-waiting-time|100s|Used when the worker disconnect from registry, and the disconnect strategy is waiting, this config means the worker will waiting to reconnect to registry in given times, and after the waiting times, if the worker still cannot connect to registry, will stop itself, if the value is 0s, will waitting infinitely |
|
||||
|worker.task-execute-threads-full-policy|REJECT|If REJECT, when the task waiting in the worker reaches exec-threads, it will reject the received task and the Master will redispatch it; If CONTINUE, it will put the task into the worker's execution queue and wait for a free thread to start execution|
|
||||
|
||||
### Alert Server related configuration
|
||||
|
||||
Location: `alert-server/conf/application.yaml`
|
||||
|
||||
|Parameters | Default value| Description|
|
||||
|
|
@ -305,7 +307,6 @@ Location: `alert-server/conf/application.yaml`
|
|||
|server.port|50053|the port of Alert Server|
|
||||
|alert.port|50052|the port of alert|
|
||||
|
||||
|
||||
### Quartz related configuration
|
||||
|
||||
This part describes quartz configs and configure them based on your practical situation and resources.
|
||||
|
|
@ -335,7 +336,6 @@ The default configuration is as follows:
|
|||
|spring.quartz.properties.org.quartz.jobStore.driverDelegateClass | org.quartz.impl.jdbcjobstore.PostgreSQLDelegate|
|
||||
|spring.quartz.properties.org.quartz.jobStore.clusterCheckinInterval | 5000|
|
||||
|
||||
|
||||
### dolphinscheduler_env.sh [load environment variables configs]
|
||||
|
||||
When using shell to commit tasks, DolphinScheduler will export environment variables from `bin/env/dolphinscheduler_env.sh`. The
|
||||
|
|
|
|||
|
|
@ -84,6 +84,7 @@
|
|||
##### Centralized Thinking
|
||||
|
||||
The centralized design concept is relatively simple. The nodes in the distributed cluster are roughly divided into two roles according to responsibilities:
|
||||
|
||||
<p align="center">
|
||||
<img src="https://analysys.github.io/easyscheduler_docs_cn/images/master_slave.png" alt="master-slave character" width="50%" />
|
||||
</p>
|
||||
|
|
@ -120,8 +121,6 @@ The service fault-tolerance design relies on ZooKeeper's Watcher mechanism, and
|
|||
</p>
|
||||
Among them, the Master monitors the directories of other Masters and Workers. If the remove event is triggered, perform fault tolerance of the process instance or task instance according to the specific business logic.
|
||||
|
||||
|
||||
|
||||
- Master fault tolerance:
|
||||
|
||||
<p align="center">
|
||||
|
|
@ -172,13 +171,14 @@ In the early schedule design, if there is no priority design and use the fair sc
|
|||
|
||||
- According to **the priority of different process instances** prior over **priority of the same process instance** prior over **priority of tasks within the same process** prior over **tasks within the same process**, process task submission order from highest to Lowest.
|
||||
- The specific implementation is to parse the priority according to the JSON of the task instance, and then save the **process instance priority_process instance id_task priority_task id** information to the ZooKeeper task queue. When obtain from the task queue, we can get the highest priority task by comparing string.
|
||||
|
||||
- The priority of the process definition is to consider that some processes need to process before other processes. Configure the priority when the process starts or schedules. There are 5 levels in total, which are HIGHEST, HIGH, MEDIUM, LOW, and LOWEST. As shown below
|
||||
|
||||
<p align="center">
|
||||
<img src="https://user-images.githubusercontent.com/10797147/146744784-eb351b14-c94a-4ed6-8ba4-5132c2a3d116.png" alt="Process priority configuration" width="40%" />
|
||||
</p>
|
||||
|
||||
- The priority of the task is also divides into 5 levels, ordered by HIGHEST, HIGH, MEDIUM, LOW, LOWEST. As shown below:
|
||||
|
||||
<p align="center">
|
||||
<img src="https://user-images.githubusercontent.com/10797147/146744830-5eac611f-5933-4f53-a0c6-31613c283708.png" alt="Task priority configuration" width="35%" />
|
||||
</p>
|
||||
|
|
@ -188,7 +188,6 @@ In the early schedule design, if there is no priority design and use the fair sc
|
|||
- Since Web (UI) and Worker are not always on the same machine, to view the log cannot be like querying a local file. There are two options:
|
||||
- Put logs on the ES search engine.
|
||||
- Obtain remote log information through netty communication.
|
||||
|
||||
- In consideration of the lightness of DolphinScheduler as much as possible, so choose gRPC to achieve remote access to log information.
|
||||
|
||||
<p align="center">
|
||||
|
|
@ -198,10 +197,10 @@ In the early schedule design, if there is no priority design and use the fair sc
|
|||
- For details, please refer to the logback configuration of Master and Worker, as shown in the following example:
|
||||
|
||||
```xml
|
||||
<conversionRule conversionWord="messsage" converterClass="org.apache.dolphinscheduler.server.log.SensitiveDataConverter"/>
|
||||
<conversionRule conversionWord="messsage" converterClass="org.apache.dolphinscheduler.service.log.SensitiveDataConverter"/>
|
||||
<appender name="TASKLOGFILE" class="ch.qos.logback.classic.sift.SiftingAppender">
|
||||
<filter class="org.apache.dolphinscheduler.server.log.TaskLogFilter"/>
|
||||
<Discriminator class="org.apache.dolphinscheduler.server.log.TaskLogDiscriminator">
|
||||
<filter class="org.apache.dolphinscheduler.service.log.TaskLogFilter"/>
|
||||
<Discriminator class="org.apache.dolphinscheduler.service.log.TaskLogDiscriminator">
|
||||
<key>taskAppId</key>
|
||||
<logBase>${log.base}</logBase>
|
||||
</Discriminator>
|
||||
|
|
|
|||
|
|
@ -57,3 +57,4 @@ You can customise the configuration by changing the following properties in work
|
|||
|
||||
- worker.max.cpuload.avg=-1 (worker max cpuload avg, only higher than the system cpu load average, worker server can be dispatched tasks. default value -1: the number of cpu cores * 2)
|
||||
- worker.reserved.memory=0.3 (worker reserved memory, only lower than system available memory, worker server can be dispatched tasks. default value 0.3, the unit is G)
|
||||
|
||||
|
|
|
|||
|
|
@ -1,8 +1,8 @@
|
|||
# MetaData
|
||||
|
||||
## Table Schema
|
||||
see sql files in `dolphinscheduler/dolphinscheduler-dao/src/main/resources/sql`
|
||||
|
||||
see sql files in `dolphinscheduler/dolphinscheduler-dao/src/main/resources/sql`
|
||||
|
||||
---
|
||||
|
||||
|
|
@ -26,6 +26,7 @@ see sql files in `dolphinscheduler/dolphinscheduler-dao/src/main/resources/sql`
|
|||
- The `user_id` in the `t_ds_udfs` table represents the user who create the UDF, and the `user_id` in the `t_ds_relation_udfs_user` table represents a user who has permission to the UDF.
|
||||
|
||||
### Project - Tenant - ProcessDefinition - Schedule
|
||||
|
||||

|
||||
|
||||
- A project can have multiple process definitions, and each process definition belongs to only one project.
|
||||
|
|
@ -33,8 +34,10 @@ see sql files in `dolphinscheduler/dolphinscheduler-dao/src/main/resources/sql`
|
|||
- A workflow definition can have one or more schedules.
|
||||
|
||||
### Process Definition Execution
|
||||
|
||||

|
||||
|
||||
- A process definition corresponds to multiple task definitions, which are associated through `t_ds_process_task_relation` and the associated key is `code + version`. When the pre-task of the task is empty, the corresponding `pre_task_node` and `pre_task_version` are 0.
|
||||
- A process definition can have multiple process instances `t_ds_process_instance`, one process instance corresponds to one or more task instances `t_ds_task_instance`.
|
||||
- The data stored in the `t_ds_relation_process_instance` table is used to handle the case that the process definition contains sub-processes. `parent_process_instance_id` represents the id of the main process instance containing the sub-process, `process_instance_id` represents the id of the sub-process instance, `parent_task_instance_id` represents the task instance id of the sub-process node. The process instance table and the task instance table correspond to the `t_ds_process_instance` table and the `t_ds_task_instance` table, respectively.
|
||||
|
||||
|
|
|
|||
|
|
@ -6,28 +6,28 @@ All tasks in DolphinScheduler are saved in the `t_ds_process_definition` table.
|
|||
|
||||
The following shows the `t_ds_process_definition` table structure:
|
||||
|
||||
No. | field | type | description
|
||||
-------- | ---------| -------- | ---------
|
||||
1|id|int(11)|primary key
|
||||
2|name|varchar(255)|process definition name
|
||||
3|version|int(11)|process definition version
|
||||
4|release_state|tinyint(4)|release status of process definition: 0 not released, 1 released
|
||||
5|project_id|int(11)|project id
|
||||
6|user_id|int(11)|user id of the process definition
|
||||
7|process_definition_json|longtext|process definition JSON
|
||||
8|description|text|process definition description
|
||||
9|global_params|text|global parameters
|
||||
10|flag|tinyint(4)|specify whether the process is available: 0 is not available, 1 is available
|
||||
11|locations|text|node location information
|
||||
12|connects|text|node connectivity info
|
||||
13|receivers|text|receivers
|
||||
14|receivers_cc|text|CC receivers
|
||||
15|create_time|datetime|create time
|
||||
16|timeout|int(11) |timeout
|
||||
17|tenant_id|int(11) |tenant id
|
||||
18|update_time|datetime|update time
|
||||
19|modify_by|varchar(36)|specify the user that made the modification
|
||||
20|resource_ids|varchar(255)|resource ids
|
||||
| No. | field | type | description |
|
||||
|-----|-------------------------|--------------|------------------------------------------------------------------------------|
|
||||
| 1 | id | int(11) | primary key |
|
||||
| 2 | name | varchar(255) | process definition name |
|
||||
| 3 | version | int(11) | process definition version |
|
||||
| 4 | release_state | tinyint(4) | release status of process definition: 0 not released, 1 released |
|
||||
| 5 | project_id | int(11) | project id |
|
||||
| 6 | user_id | int(11) | user id of the process definition |
|
||||
| 7 | process_definition_json | longtext | process definition JSON |
|
||||
| 8 | description | text | process definition description |
|
||||
| 9 | global_params | text | global parameters |
|
||||
| 10 | flag | tinyint(4) | specify whether the process is available: 0 is not available, 1 is available |
|
||||
| 11 | locations | text | node location information |
|
||||
| 12 | connects | text | node connectivity info |
|
||||
| 13 | receivers | text | receivers |
|
||||
| 14 | receivers_cc | text | CC receivers |
|
||||
| 15 | create_time | datetime | create time |
|
||||
| 16 | timeout | int(11) | timeout |
|
||||
| 17 | tenant_id | int(11) | tenant id |
|
||||
| 18 | update_time | datetime | update time |
|
||||
| 19 | modify_by | varchar(36) | specify the user that made the modification |
|
||||
| 20 | resource_ids | varchar(255) | resource ids |
|
||||
|
||||
The `process_definition_json` field is the core field, which defines the task information in the DAG diagram, and it is stored in JSON format.
|
||||
|
||||
|
|
@ -40,6 +40,7 @@ No. | field | type | description
|
|||
4|timeout|int|timeout
|
||||
|
||||
Data example:
|
||||
|
||||
```bash
|
||||
{
|
||||
"globalParams":[
|
||||
|
|
@ -238,38 +239,38 @@ No.|parameter name||type|description |note
|
|||
|
||||
**The following shows the node data structure:**
|
||||
|
||||
No.|parameter name||type|description |notes
|
||||
-------- | ---------| ---------| -------- | --------- | ---------
|
||||
1|id | |String| task Id|
|
||||
2|type ||String |task type |SPARK
|
||||
3| name| |String|task name |
|
||||
4| params| |Object|customized parameters |JSON format
|
||||
5| |mainClass |String | main class
|
||||
6| |mainArgs | String| execution arguments
|
||||
7| |others | String| other arguments
|
||||
8| |mainJar |Object | application jar package
|
||||
9| |deployMode |String |deployment mode |local,client,cluster
|
||||
10| |driverCores | String| driver cores
|
||||
11| |driverMemory | String| driver memory
|
||||
12| |numExecutors |String | executor count
|
||||
13| |executorMemory |String | executor memory
|
||||
14| |executorCores |String | executor cores
|
||||
15| |programType | String| program type|JAVA,SCALA,PYTHON
|
||||
16| | sparkVersion| String| Spark version| SPARK1 , SPARK2
|
||||
17| | localParams| Array|customized local parameters
|
||||
18| | resourceList| Array|resource files
|
||||
19|description | |String|description | |
|
||||
20|runFlag | |String |execution flag| |
|
||||
21|conditionResult | |Object|condition branch| |
|
||||
22| | successNode| Array|jump to node if success| |
|
||||
23| | failedNode|Array|jump to node if failure|
|
||||
24| dependence| |Object |task dependency |mutual exclusion with params
|
||||
25|maxRetryTimes | |String|max retry times | |
|
||||
26|retryInterval | |String |retry interval| |
|
||||
27|timeout | |Object|timeout | |
|
||||
28| taskInstancePriority| |String|task priority | |
|
||||
29|workerGroup | |String |Worker group| |
|
||||
30|preTasks | |Array|preposition tasks| |
|
||||
| No. | parameter name || type | description | notes |
|
||||
|-----|----------------------|----------------|--------|-----------------------------|------------------------------|
|
||||
| 1 | id | | String | task Id |
|
||||
| 2 | type || String | task type | SPARK |
|
||||
| 3 | name | | String | task name |
|
||||
| 4 | params | | Object | customized parameters | JSON format |
|
||||
| 5 | | mainClass | String | main class |
|
||||
| 6 | | mainArgs | String | execution arguments |
|
||||
| 7 | | others | String | other arguments |
|
||||
| 8 | | mainJar | Object | application jar package |
|
||||
| 9 | | deployMode | String | deployment mode | local,client,cluster |
|
||||
| 10 | | driverCores | String | driver cores |
|
||||
| 11 | | driverMemory | String | driver memory |
|
||||
| 12 | | numExecutors | String | executor count |
|
||||
| 13 | | executorMemory | String | executor memory |
|
||||
| 14 | | executorCores | String | executor cores |
|
||||
| 15 | | programType | String | program type | JAVA,SCALA,PYTHON |
|
||||
| 16 | | sparkVersion | String | Spark version | SPARK1 , SPARK2 |
|
||||
| 17 | | localParams | Array | customized local parameters |
|
||||
| 18 | | resourceList | Array | resource files |
|
||||
| 19 | description | | String | description | |
|
||||
| 20 | runFlag | | String | execution flag | |
|
||||
| 21 | conditionResult | | Object | condition branch | |
|
||||
| 22 | | successNode | Array | jump to node if success | |
|
||||
| 23 | | failedNode | Array | jump to node if failure |
|
||||
| 24 | dependence | | Object | task dependency | mutual exclusion with params |
|
||||
| 25 | maxRetryTimes | | String | max retry times | |
|
||||
| 26 | retryInterval | | String | retry interval | |
|
||||
| 27 | timeout | | Object | timeout | |
|
||||
| 28 | taskInstancePriority | | String | task priority | |
|
||||
| 29 | workerGroup | | String | Worker group | |
|
||||
| 30 | preTasks | | Array | preposition tasks | |
|
||||
|
||||
**Node data example:**
|
||||
|
||||
|
|
@ -336,31 +337,31 @@ No.|parameter name||type|description |notes
|
|||
|
||||
**The following shows the node data structure:**
|
||||
|
||||
No.|parameter name||type|description |notes
|
||||
-------- | ---------| ---------| -------- | --------- | ---------
|
||||
1|id | |String| task Id|
|
||||
2|type ||String |task type |MR
|
||||
3| name| |String|task name |
|
||||
4| params| |Object|customized parameters |JSON format
|
||||
5| |mainClass |String | main class
|
||||
6| |mainArgs | String|execution arguments
|
||||
7| |others | String|other arguments
|
||||
8| |mainJar |Object | application jar package
|
||||
9| |programType | String|program type|JAVA,PYTHON
|
||||
10| | localParams| Array|customized local parameters
|
||||
11| | resourceList| Array|resource files
|
||||
12|description | |String|description | |
|
||||
13|runFlag | |String |execution flag| |
|
||||
14|conditionResult | |Object|condition branch| |
|
||||
15| | successNode| Array|jump to node if success| |
|
||||
16| | failedNode|Array|jump to node if failure|
|
||||
17| dependence| |Object |task dependency |mutual exclusion with params
|
||||
18|maxRetryTimes | |String|max retry times | |
|
||||
19|retryInterval | |String |retry interval| |
|
||||
20|timeout | |Object|timeout | |
|
||||
21| taskInstancePriority| |String|task priority| |
|
||||
22|workerGroup | |String |Worker group| |
|
||||
23|preTasks | |Array|preposition tasks| |
|
||||
| No. | parameter name || type | description | notes |
|
||||
|-----|----------------------|--------------|--------|-----------------------------|------------------------------|
|
||||
| 1 | id | | String | task Id |
|
||||
| 2 | type || String | task type | MR |
|
||||
| 3 | name | | String | task name |
|
||||
| 4 | params | | Object | customized parameters | JSON format |
|
||||
| 5 | | mainClass | String | main class |
|
||||
| 6 | | mainArgs | String | execution arguments |
|
||||
| 7 | | others | String | other arguments |
|
||||
| 8 | | mainJar | Object | application jar package |
|
||||
| 9 | | programType | String | program type | JAVA,PYTHON |
|
||||
| 10 | | localParams | Array | customized local parameters |
|
||||
| 11 | | resourceList | Array | resource files |
|
||||
| 12 | description | | String | description | |
|
||||
| 13 | runFlag | | String | execution flag | |
|
||||
| 14 | conditionResult | | Object | condition branch | |
|
||||
| 15 | | successNode | Array | jump to node if success | |
|
||||
| 16 | | failedNode | Array | jump to node if failure |
|
||||
| 17 | dependence | | Object | task dependency | mutual exclusion with params |
|
||||
| 18 | maxRetryTimes | | String | max retry times | |
|
||||
| 19 | retryInterval | | String | retry interval | |
|
||||
| 20 | timeout | | Object | timeout | |
|
||||
| 21 | taskInstancePriority | | String | task priority | |
|
||||
| 22 | workerGroup | | String | Worker group | |
|
||||
| 23 | preTasks | | Array | preposition tasks | |
|
||||
|
||||
**Node data example:**
|
||||
|
||||
|
|
@ -493,36 +494,36 @@ No.|parameter name||type|description |notes
|
|||
|
||||
**The following shows the node data structure:**
|
||||
|
||||
No.|parameter name||type|description |notes
|
||||
-------- | ---------| ---------| -------- | --------- | ---------
|
||||
1|id | |String|task Id|
|
||||
2|type ||String |task type|FLINK
|
||||
3| name| |String|task name|
|
||||
4| params| |Object|customized parameters |JSON format
|
||||
5| |mainClass |String |main class
|
||||
6| |mainArgs | String|execution arguments
|
||||
7| |others | String|other arguments
|
||||
8| |mainJar |Object |application jar package
|
||||
9| |deployMode |String |deployment mode |local,client,cluster
|
||||
10| |slot | String| slot count
|
||||
11| |taskManager |String | taskManager count
|
||||
12| |taskManagerMemory |String |taskManager memory size
|
||||
13| |jobManagerMemory |String | jobManager memory size
|
||||
14| |programType | String| program type|JAVA,SCALA,PYTHON
|
||||
15| | localParams| Array|local parameters
|
||||
16| | resourceList| Array|resource files
|
||||
17|description | |String|description | |
|
||||
18|runFlag | |String |execution flag| |
|
||||
19|conditionResult | |Object|condition branch| |
|
||||
20| | successNode| Array|jump node if success| |
|
||||
21| | failedNode|Array|jump node if failure|
|
||||
22| dependence| |Object |task dependency |mutual exclusion with params
|
||||
23|maxRetryTimes | |String|max retry times| |
|
||||
24|retryInterval | |String |retry interval| |
|
||||
25|timeout | |Object|timeout | |
|
||||
26| taskInstancePriority| |String|task priority| |
|
||||
27|workerGroup | |String |Worker group| |
|
||||
38|preTasks | |Array|preposition tasks| |
|
||||
| No. | parameter name || type | description | notes |
|
||||
|-----|----------------------|-------------------|--------|-------------------------|------------------------------|
|
||||
| 1 | id | | String | task Id |
|
||||
| 2 | type || String | task type | FLINK |
|
||||
| 3 | name | | String | task name |
|
||||
| 4 | params | | Object | customized parameters | JSON format |
|
||||
| 5 | | mainClass | String | main class |
|
||||
| 6 | | mainArgs | String | execution arguments |
|
||||
| 7 | | others | String | other arguments |
|
||||
| 8 | | mainJar | Object | application jar package |
|
||||
| 9 | | deployMode | String | deployment mode | local,client,cluster |
|
||||
| 10 | | slot | String | slot count |
|
||||
| 11 | | taskManager | String | taskManager count |
|
||||
| 12 | | taskManagerMemory | String | taskManager memory size |
|
||||
| 13 | | jobManagerMemory | String | jobManager memory size |
|
||||
| 14 | | programType | String | program type | JAVA,SCALA,PYTHON |
|
||||
| 15 | | localParams | Array | local parameters |
|
||||
| 16 | | resourceList | Array | resource files |
|
||||
| 17 | description | | String | description | |
|
||||
| 18 | runFlag | | String | execution flag | |
|
||||
| 19 | conditionResult | | Object | condition branch | |
|
||||
| 20 | | successNode | Array | jump node if success | |
|
||||
| 21 | | failedNode | Array | jump node if failure |
|
||||
| 22 | dependence | | Object | task dependency | mutual exclusion with params |
|
||||
| 23 | maxRetryTimes | | String | max retry times | |
|
||||
| 24 | retryInterval | | String | retry interval | |
|
||||
| 25 | timeout | | Object | timeout | |
|
||||
| 26 | taskInstancePriority | | String | task priority | |
|
||||
| 27 | workerGroup | | String | Worker group | |
|
||||
| 38 | preTasks | | Array | preposition tasks | |
|
||||
|
||||
**Node data example:**
|
||||
|
||||
|
|
@ -588,30 +589,30 @@ No.|parameter name||type|description |notes
|
|||
|
||||
**The following shows the node data structure:**
|
||||
|
||||
No.|parameter name||type|description |notes
|
||||
-------- | ---------| ---------| -------- | --------- | ---------
|
||||
1|id | |String|task Id|
|
||||
2|type ||String |task type|HTTP
|
||||
3| name| |String|task name|
|
||||
4| params| |Object|customized parameters |JSON format
|
||||
5| |url |String |request url
|
||||
6| |httpMethod | String|http method|GET,POST,HEAD,PUT,DELETE
|
||||
7| | httpParams| Array|http parameters
|
||||
8| |httpCheckCondition | String|validation of HTTP code status|default code 200
|
||||
9| |condition |String |validation conditions
|
||||
10| | localParams| Array|customized local parameters
|
||||
11|description | |String|description| |
|
||||
12|runFlag | |String |execution flag| |
|
||||
13|conditionResult | |Object|condition branch| |
|
||||
14| | successNode| Array|jump node if success| |
|
||||
15| | failedNode|Array|jump node if failure|
|
||||
16| dependence| |Object |task dependency |mutual exclusion with params
|
||||
17|maxRetryTimes | |String|max retry times | |
|
||||
18|retryInterval | |String |retry interval| |
|
||||
19|timeout | |Object|timeout | |
|
||||
20| taskInstancePriority| |String|task priority| |
|
||||
21|workerGroup | |String |Worker group| |
|
||||
22|preTasks | |Array|preposition tasks| |
|
||||
| No. | parameter name || type | description | notes |
|
||||
|-----|----------------------|--------------------|--------|--------------------------------|------------------------------|
|
||||
| 1 | id | | String | task Id |
|
||||
| 2 | type || String | task type | HTTP |
|
||||
| 3 | name | | String | task name |
|
||||
| 4 | params | | Object | customized parameters | JSON format |
|
||||
| 5 | | url | String | request url |
|
||||
| 6 | | httpMethod | String | http method | GET,POST,HEAD,PUT,DELETE |
|
||||
| 7 | | httpParams | Array | http parameters |
|
||||
| 8 | | httpCheckCondition | String | validation of HTTP code status | default code 200 |
|
||||
| 9 | | condition | String | validation conditions |
|
||||
| 10 | | localParams | Array | customized local parameters |
|
||||
| 11 | description | | String | description | |
|
||||
| 12 | runFlag | | String | execution flag | |
|
||||
| 13 | conditionResult | | Object | condition branch | |
|
||||
| 14 | | successNode | Array | jump node if success | |
|
||||
| 15 | | failedNode | Array | jump node if failure |
|
||||
| 16 | dependence | | Object | task dependency | mutual exclusion with params |
|
||||
| 17 | maxRetryTimes | | String | max retry times | |
|
||||
| 18 | retryInterval | | String | retry interval | |
|
||||
| 19 | timeout | | Object | timeout | |
|
||||
| 20 | taskInstancePriority | | String | task priority | |
|
||||
| 21 | workerGroup | | String | Worker group | |
|
||||
| 22 | preTasks | | Array | preposition tasks | |
|
||||
|
||||
**Node data example:**
|
||||
|
||||
|
|
@ -1112,3 +1113,4 @@ No.|parameter name||type|description |notes
|
|||
]
|
||||
}
|
||||
```
|
||||
|
||||
|
|
|
|||
|
|
@ -0,0 +1,66 @@
|
|||
# Change Log
|
||||
|
||||
## Bugfix
|
||||
|
||||
- [fix-12675]edit workflow related task, workflow's task version change
|
||||
- [Bug] Resource default auth function disabled false.
|
||||
- [3.1.1][api]Fix updating workflow definition causing task definition data to be duplicated. (#12712)
|
||||
- Fix flink sql cannot run due to missing main jar (#12705)
|
||||
- Add pythonNodePort in config file (#12685)
|
||||
- [Fix-12109]Fix the errors when starting 2 times with dolphinscheduler-daemon.sh (#12118)
|
||||
- Fix the waiting strategy cannot recovery if the serverstate is already in running (#12651)
|
||||
- Fix alert status spelling error #12592
|
||||
- Add configmap resource permissions so config hot reload can work (#12572)
|
||||
- [Fix][ui] download resource return 401 (#12566)
|
||||
- [Bug] [API] Before deleting a worker group, check whether there is environment that reference the worker group. #12534
|
||||
- [Fix-12518][swagger]Fill up missing i18 properties (#12599)
|
||||
- [Bug][master] Add the aws-java-sdk-s3 jar package to the master module (#12259) (#12512)
|
||||
- Remove equals in User to fix UT #12487
|
||||
- [Bug] Set tenantDir permission #12486
|
||||
- [Fix-12451][k8s] Read the kubeconfig from cluster conf (#12452)
|
||||
- [Fix-12356][k8s] fix the null exception when submitting k8s task plugin (#12358)
|
||||
- [fix#12439] [Alert] fix send script alert NPE #12495
|
||||
- [Bug] [API] The workflow definition and the tenant in the workflow instance are inconsistent. (#12533)
|
||||
- [fix](dolphinscheduler-dao) fix upgrade to 3.1.0 sql missing field (#12314) (#12315)
|
||||
- [Fix-12425][api] Add rollbackFor setting.
|
||||
- [Bug-12410] [API]Fix the worker list result in workflow definition only has default
|
||||
- [BUG-12306][ui]Fix the password item always is disabled (#12437)
|
||||
- [Fix][task] Fix dependent task can not predicate the status of the corresponding task correctly (#12253)
|
||||
- make sure all failed task will save in errorTaskMap (#12424)
|
||||
- Fix timing scheduling trigger master service report to get command parameter null pointer exception (#12419)
|
||||
- [Bug] [spark-sql] In spark-sql, select both SPARK1 and SPARK2 versions and execute ${SPARK_HOME2}/bin/spark-sql (#11721) (#12420)
|
||||
- fix hdfs defaultFs not working (#11823) (#12418)
|
||||
- source is not available in sh (#12413)
|
||||
- fix datax NPE issue (#12388) (#12407)
|
||||
- [BUG-12396][schedule] Fixed that the workflow definition scheduling that has been online after the version upgrade does not execute
|
||||
|
||||
## Doc
|
||||
|
||||
- [doc] Correct descriptions in glossary.md (#12282)
|
||||
- [Hotfix][docs] Fix 404 dead link
|
||||
- update english oracle.md (#12332)
|
||||
|
||||
## Improvement
|
||||
|
||||
- adjust the args of router in the dag (#12759)
|
||||
- Change command file permission to 755 (#12678)
|
||||
- beautify the dag (#12728)
|
||||
- support to use the clearable button of components to search (#12668)
|
||||
- Add worker-group-refresh-interval in master config #12601
|
||||
- [Improvement][ui] Support to view the process variables on the page of DAG. (#12609)
|
||||
- [Improvement] Merge spi.utils into common.utils (#12607)
|
||||
- Add task executor threads full policy config in worker (#12510)
|
||||
- Add mysql support to helm chart (#12517)
|
||||
- Reorganize some classes in common module, remove duplicate classes #12321
|
||||
- [DS-12131][master] Optimize the log printing of the master module acc… (#12152)
|
||||
- [Improvement][workergroup]Remove workerGroup in registry #12217
|
||||
- [Improvement] remove log-server and server module #12206
|
||||
- Refactor LogServiceClient Singleton to avoid repeat creation of NettyClient #11777
|
||||
- [Improvement] Add remote task model #11767 (#12541)
|
||||
- Use temurin Java docker images instead of deprecated ones (#12334)
|
||||
- [DS-12154][worker] Optimize the log printing of the worker module (#12183)
|
||||
- [Improvement-12372][k8s] Update the deprecated k8s api (#12373)
|
||||
- [Improvement][task plugin] Modify the comment of 'deployMode'. (#12163)
|
||||
- Remove the DataxTaskTest class because there is no junit5 package.
|
||||
- [Improvement][api] When the workflow definition is copied, the operation user of the timed instance is changed to the current user
|
||||
- [Improvement-12391][api] Workflow definitions that contain logical task nodes support the copy function
|
||||
|
|
@ -1,9 +1,11 @@
|
|||
# API design standard
|
||||
|
||||
A standardized and unified API is the cornerstone of project design.The API of DolphinScheduler follows the REST ful standard. REST ful is currently the most popular Internet software architecture. It has a clear structure, conforms to standards, is easy to understand and extend.
|
||||
|
||||
This article uses the DolphinScheduler API as an example to explain how to construct a Restful API.
|
||||
|
||||
## 1. URI design
|
||||
|
||||
REST is "Representational State Transfer".The design of Restful URI is based on resources.The resource corresponds to an entity on the network, for example: a piece of text, a picture, and a service. And each resource corresponds to a URI.
|
||||
|
||||
+ One Kind of Resource: expressed in the plural, such as `task-instances`、`groups` ;
|
||||
|
|
@ -12,36 +14,43 @@ REST is "Representational State Transfer".The design of Restful URI is based on
|
|||
+ A Sub Resource:`/instances/{instanceId}/tasks/{taskId}`;
|
||||
|
||||
## 2. Method design
|
||||
|
||||
We need to locate a certain resource by URI, and then use Method or declare actions in the path suffix to reflect the operation of the resource.
|
||||
|
||||
### ① Query - GET
|
||||
|
||||
Use URI to locate the resource, and use GET to indicate query.
|
||||
|
||||
+ When the URI is a type of resource, it means to query a type of resource. For example, the following example indicates paging query `alter-groups`.
|
||||
|
||||
```
|
||||
Method: GET
|
||||
/dolphinscheduler/alert-groups
|
||||
```
|
||||
|
||||
+ When the URI is a single resource, it means to query this resource. For example, the following example means to query the specified `alter-group`.
|
||||
|
||||
```
|
||||
Method: GET
|
||||
/dolphinscheduler/alter-groups/{id}
|
||||
```
|
||||
|
||||
+ In addition, we can also express query sub-resources based on URI, as follows:
|
||||
|
||||
```
|
||||
Method: GET
|
||||
/dolphinscheduler/projects/{projectId}/tasks
|
||||
```
|
||||
|
||||
**The above examples all represent paging query. If we need to query all data, we need to add `/list` after the URI to distinguish. Do not mix the same API for both paged query and query.**
|
||||
|
||||
```
|
||||
Method: GET
|
||||
/dolphinscheduler/alert-groups/list
|
||||
```
|
||||
|
||||
### ② Create - POST
|
||||
|
||||
Use URI to locate the resource, use POST to indicate create, and then return the created id to requester.
|
||||
|
||||
+ create an `alter-group`:
|
||||
|
|
@ -52,35 +61,42 @@ Method: POST
|
|||
```
|
||||
|
||||
+ create sub-resources is also the same as above.
|
||||
|
||||
```
|
||||
Method: POST
|
||||
/dolphinscheduler/alter-groups/{alterGroupId}/tasks
|
||||
```
|
||||
|
||||
### ③ Modify - PUT
|
||||
|
||||
Use URI to locate the resource, use PUT to indicate modify.
|
||||
+ modify an `alert-group`
|
||||
|
||||
```
|
||||
Method: PUT
|
||||
/dolphinscheduler/alter-groups/{alterGroupId}
|
||||
```
|
||||
|
||||
### ④ Delete -DELETE
|
||||
|
||||
Use URI to locate the resource, use DELETE to indicate delete.
|
||||
|
||||
+ delete an `alert-group`
|
||||
|
||||
```
|
||||
Method: DELETE
|
||||
/dolphinscheduler/alter-groups/{alterGroupId}
|
||||
```
|
||||
|
||||
+ batch deletion: batch delete the id array,we should use POST. **(Do not use the DELETE method, because the body of the DELETE request has no semantic meaning, and it is possible that some gateways, proxies, and firewalls will directly strip off the request body after receiving the DELETE request.)**
|
||||
|
||||
```
|
||||
Method: POST
|
||||
/dolphinscheduler/alter-groups/batch-delete
|
||||
```
|
||||
|
||||
### ⑤ Partial Modifications -PATCH
|
||||
|
||||
Use URI to locate the resource, use PATCH to partial modifications.
|
||||
|
||||
```
|
||||
|
|
@ -89,20 +105,27 @@ Method: PATCH
|
|||
```
|
||||
|
||||
### ⑥ Others
|
||||
|
||||
In addition to creating, deleting, modifying and quering, we also locate the corresponding resource through url, and then append operations to it after the path, such as:
|
||||
|
||||
```
|
||||
/dolphinscheduler/alert-groups/verify-name
|
||||
/dolphinscheduler/projects/{projectCode}/process-instances/{code}/view-gantt
|
||||
```
|
||||
|
||||
## 3. Parameter design
|
||||
|
||||
There are two types of parameters, one is request parameter and the other is path parameter. And the parameter must use small hump.
|
||||
|
||||
In the case of paging, if the parameter entered by the user is less than 1, the front end needs to automatically turn to 1, indicating that the first page is requested; When the backend finds that the parameter entered by the user is greater than the total number of pages, it should directly return to the last page.
|
||||
|
||||
## 4. Others design
|
||||
|
||||
### base URL
|
||||
|
||||
The URI of the project needs to use `/<project_name>` as the base path, so as to identify that these APIs are under this project.
|
||||
|
||||
```
|
||||
/dolphinscheduler
|
||||
```
|
||||
|
||||
|
|
|
|||
|
|
@ -10,7 +10,6 @@ In contrast, API testing focuses on whether a complete operation chain can be co
|
|||
|
||||
For example, the API test of the tenant management interface focuses on whether users can log in normally; If the login fails, whether the error message can be displayed correctly. After logging in, you can perform tenant management operations through the sessionid you carry.
|
||||
|
||||
|
||||
## API Test
|
||||
|
||||
### API-Pages
|
||||
|
|
@ -49,7 +48,6 @@ In addition, during the testing process, the interface are not requested directl
|
|||
|
||||
On the login page, only the input parameter specification of the interface request is defined. For the output parameter of the interface request, only the unified basic response structure is defined. The data actually returned by the interface is tested in the actual test case. Whether the input and output of main test interfaces can meet the requirements of test cases.
|
||||
|
||||
|
||||
### API-Cases
|
||||
|
||||
The following is an example of a tenant management test. As explained earlier, we use docker-compose for deployment, so for each test case, we need to import the corresponding file in the form of an annotation.
|
||||
|
|
|
|||
|
|
@ -1,4 +1,5 @@
|
|||
## Architecture Design
|
||||
|
||||
Before explaining the architecture of the schedule system, let us first understand the common nouns of the schedule system.
|
||||
|
||||
### 1.Noun Interpretation
|
||||
|
|
@ -34,11 +35,10 @@ Before explaining the architecture of the schedule system, let us first understa
|
|||
|
||||
**Complement**: Complement historical data, support **interval parallel and serial** two complement methods
|
||||
|
||||
|
||||
|
||||
### 2.System architecture
|
||||
|
||||
#### 2.1 System Architecture Diagram
|
||||
|
||||
<p align="center">
|
||||
<img src="../../../img/architecture.jpg" alt="System Architecture Diagram" />
|
||||
<p align="center">
|
||||
|
|
@ -46,8 +46,6 @@ Before explaining the architecture of the schedule system, let us first understa
|
|||
</p>
|
||||
</p>
|
||||
|
||||
|
||||
|
||||
#### 2.2 Architectural description
|
||||
|
||||
* **MasterServer**
|
||||
|
|
@ -55,8 +53,6 @@ Before explaining the architecture of the schedule system, let us first understa
|
|||
MasterServer adopts the distributed non-central design concept. MasterServer is mainly responsible for DAG task split, task submission monitoring, and monitoring the health status of other MasterServer and WorkerServer.
|
||||
When the MasterServer service starts, it registers a temporary node with Zookeeper, and listens to the Zookeeper temporary node state change for fault tolerance processing.
|
||||
|
||||
|
||||
|
||||
##### The service mainly contains:
|
||||
|
||||
- **Distributed Quartz** distributed scheduling component, mainly responsible for the start and stop operation of the scheduled task. When the quartz picks up the task, the master internally has a thread pool to be responsible for the subsequent operations of the task.
|
||||
|
|
@ -67,8 +63,6 @@ Before explaining the architecture of the schedule system, let us first understa
|
|||
|
||||
- **MasterTaskExecThread** is mainly responsible for task persistence
|
||||
|
||||
|
||||
|
||||
* **WorkerServer**
|
||||
|
||||
- WorkerServer also adopts a distributed, non-central design concept. WorkerServer is mainly responsible for task execution and providing log services. When the WorkerServer service starts, it registers the temporary node with Zookeeper and maintains the heartbeat.
|
||||
|
|
@ -76,7 +70,6 @@ Before explaining the architecture of the schedule system, let us first understa
|
|||
##### This service contains:
|
||||
|
||||
- **FetchTaskThread** is mainly responsible for continuously receiving tasks from **Task Queue** and calling **TaskScheduleThread** corresponding executors according to different task types.
|
||||
|
||||
- **ZooKeeper**
|
||||
|
||||
The ZooKeeper service, the MasterServer and the WorkerServer nodes in the system all use the ZooKeeper for cluster management and fault tolerance. In addition, the system also performs event monitoring and distributed locking based on ZooKeeper.
|
||||
|
|
@ -99,8 +92,6 @@ Before explaining the architecture of the schedule system, let us first understa
|
|||
|
||||
The front-end page of the system provides various visual operation interfaces of the system. For details, see the [quick start](https://dolphinscheduler.apache.org/en-us/docs/latest/user_doc/about/introduction.html) section.
|
||||
|
||||
|
||||
|
||||
#### 2.3 Architectural Design Ideas
|
||||
|
||||
##### I. Decentralized vs centralization
|
||||
|
|
@ -130,7 +121,6 @@ Problems in the design of centralized :
|
|||
- In the decentralized design, there is usually no Master/Slave concept, all roles are the same, the status is equal, the global Internet is a typical decentralized distributed system, networked arbitrary node equipment down machine , all will only affect a small range of features.
|
||||
- The core design of decentralized design is that there is no "manager" that is different from other nodes in the entire distributed system, so there is no single point of failure problem. However, since there is no "manager" node, each node needs to communicate with other nodes to get the necessary machine information, and the unreliable line of distributed system communication greatly increases the difficulty of implementing the above functions.
|
||||
- In fact, truly decentralized distributed systems are rare. Instead, dynamic centralized distributed systems are constantly emerging. Under this architecture, the managers in the cluster are dynamically selected, rather than preset, and when the cluster fails, the nodes of the cluster will spontaneously hold "meetings" to elect new "managers". Go to preside over the work. The most typical case is the Etcd implemented in ZooKeeper and Go.
|
||||
|
||||
- Decentralization of DolphinScheduler is the registration of Master/Worker to ZooKeeper. The Master Cluster and the Worker Cluster are not centered, and the Zookeeper distributed lock is used to elect one Master or Worker as the “manager” to perform the task.
|
||||
|
||||
##### 二、Distributed lock practice
|
||||
|
|
@ -184,8 +174,6 @@ Service fault tolerance design relies on ZooKeeper's Watcher mechanism. The impl
|
|||
|
||||
The Master monitors the directories of other Masters and Workers. If the remove event is detected, the process instance is fault-tolerant or the task instance is fault-tolerant according to the specific business logic.
|
||||
|
||||
|
||||
|
||||
- Master fault tolerance flow chart:
|
||||
|
||||
<p align="center">
|
||||
|
|
@ -194,8 +182,6 @@ The Master monitors the directories of other Masters and Workers. If the remove
|
|||
|
||||
After the ZooKeeper Master is fault-tolerant, it is rescheduled by the Scheduler thread in DolphinScheduler. It traverses the DAG to find the "Running" and "Submit Successful" tasks, and monitors the status of its task instance for the "Running" task. You need to determine whether the Task Queue already exists. If it exists, monitor the status of the task instance. If it does not exist, resubmit the task instance.
|
||||
|
||||
|
||||
|
||||
- Worker fault tolerance flow chart:
|
||||
|
||||
<p align="center">
|
||||
|
|
@ -214,8 +200,6 @@ Here we must first distinguish between the concept of task failure retry, proces
|
|||
- Process failure recovery is process level, is done manually, recovery can only be performed **from the failed node** or **from the current node**
|
||||
- Process failure rerun is also process level, is done manually, rerun is from the start node
|
||||
|
||||
|
||||
|
||||
Next, let's talk about the topic, we divided the task nodes in the workflow into two types.
|
||||
|
||||
- One is a business node, which corresponds to an actual script or processing statement, such as a Shell node, an MR node, a Spark node, a dependent node, and so on.
|
||||
|
|
@ -225,16 +209,12 @@ Each **service node** can configure the number of failed retries. When the task
|
|||
|
||||
If there is a task failure in the workflow that reaches the maximum number of retries, the workflow will fail to stop, and the failed workflow can be manually rerun or process resumed.
|
||||
|
||||
|
||||
|
||||
##### V. Task priority design
|
||||
|
||||
In the early scheduling design, if there is no priority design and fair scheduling design, it will encounter the situation that the task submitted first may be completed simultaneously with the task submitted subsequently, but the priority of the process or task cannot be set. We have redesigned this, and we are currently designing it as follows:
|
||||
|
||||
- According to **different process instance priority** prioritizes **same process instance priority** prioritizes **task priority within the same process** takes precedence over **same process** commit order from high Go to low for task processing.
|
||||
|
||||
- The specific implementation is to resolve the priority according to the json of the task instance, and then save the **process instance priority _ process instance id_task priority _ task id** information in the ZooKeeper task queue, when obtained from the task queue, Through string comparison, you can get the task that needs to be executed first.
|
||||
|
||||
- The priority of the process definition is that some processes need to be processed before other processes. This can be configured at the start of the process or at the time of scheduled start. There are 5 levels, followed by HIGHEST, HIGH, MEDIUM, LOW, and LOWEST. As shown below
|
||||
|
||||
<p align="center">
|
||||
|
|
@ -308,8 +288,6 @@ Public class TaskLogFilter extends Filter<ILoggingEvent> {
|
|||
}
|
||||
```
|
||||
|
||||
|
||||
|
||||
### summary
|
||||
|
||||
Starting from the scheduling, this paper introduces the architecture principle and implementation ideas of the big data distributed workflow scheduling system-DolphinScheduler. To be continued
|
||||
|
|
|
|||
|
|
@ -59,3 +59,4 @@ Assign the parameters with matching values to varPool (List, which contains the
|
|||
|
||||
* Format the varPool as json and pass it to master.
|
||||
* The parameters that are OUT would be written into the localParam after the master has received the varPool.
|
||||
|
||||
|
|
|
|||
|
|
@ -1,6 +1,6 @@
|
|||
# Overview
|
||||
|
||||
<!-- TODO Since the side menu does not support multiple levels, add new page to keep all sub page here -->
|
||||
|
||||
* [Global Parameter](global-parameter.md)
|
||||
* [Switch Task type](task/switch.md)
|
||||
|
||||
|
|
|
|||
|
|
@ -6,3 +6,4 @@ Switch task workflow step as follows
|
|||
* `SwitchTaskExecThread` processes the expressions defined in `switch` from top to bottom, obtains the value of the variable from `varPool`, and parses the expression through `javascript`. If the expression returns true, stop checking and record The order of the expression, here we record as resultConditionLocation. The task of SwitchTaskExecThread is over
|
||||
* After the `switch` task runs, if there is no error (more commonly, the user-defined expression is out of specification or there is a problem with the parameter name), then `MasterExecThread.submitPostNode` will obtain the downstream node of the `DAG` to continue execution.
|
||||
* If it is found in `DagHelper.parsePostNodes` that the current node (the node that has just completed the work) is a `switch` node, the `resultConditionLocation` will be obtained, and all branches except `resultConditionLocation` in the SwitchParameters will be skipped. In this way, only the branches that need to be executed are left
|
||||
|
||||
|
|
|
|||
|
|
@ -26,8 +26,8 @@ If you don't care about its internal design, but simply want to know how to deve
|
|||
|
||||
This module is currently a plug-in provided by us, and now we have supported dozens of plug-ins, such as Email, DingTalk, Script, etc.
|
||||
|
||||
|
||||
#### Alert SPI Main class information.
|
||||
|
||||
AlertChannelFactory
|
||||
Alarm plug-in factory interface. All alarm plug-ins need to implement this interface. This interface is used to define the name of the alarm plug-in and the required parameters. The create method is used to create a specific alarm plug-in instance.
|
||||
|
||||
|
|
@ -77,15 +77,19 @@ The specific design of alert_spi can be seen in the issue: [Alert Plugin Design]
|
|||
* SMS
|
||||
|
||||
SMS alerts
|
||||
|
||||
* FeiShu
|
||||
|
||||
FeiShu alert notification
|
||||
|
||||
* Slack
|
||||
|
||||
Slack alert notification
|
||||
|
||||
* PagerDuty
|
||||
|
||||
PagerDuty alert notification
|
||||
|
||||
* WebexTeams
|
||||
|
||||
WebexTeams alert notification
|
||||
|
|
@ -101,3 +105,4 @@ The specific design of alert_spi can be seen in the issue: [Alert Plugin Design]
|
|||
* Http
|
||||
|
||||
We have implemented a Http script for alerting. And calling most of the alerting plug-ins end up being Http requests, if we not support your alert plug-in yet, you can use Http to realize your alert login. Also welcome to contribute your common plug-ins to the community :)
|
||||
|
||||
|
|
|
|||
|
|
@ -6,6 +6,7 @@ Make the following configuration (take zookeeper as an example)
|
|||
|
||||
* Registry plug-in configuration, take Zookeeper as an example (registry.properties)
|
||||
dolphinscheduler-service/src/main/resources/registry.properties
|
||||
|
||||
```registry.properties
|
||||
registry.plugin.name=zookeeper
|
||||
registry.servers=127.0.0.1:2181
|
||||
|
|
|
|||
|
|
@ -1,6 +1,7 @@
|
|||
# DolphinScheduler development
|
||||
|
||||
## Software Requirements
|
||||
|
||||
Before setting up the DolphinScheduler development environment, please make sure you have installed the software as below:
|
||||
|
||||
* [Git](https://git-scm.com/downloads)
|
||||
|
|
@ -46,6 +47,7 @@ fix things for you.
|
|||
DolphinScheduler will release new Docker images after it released, you could find them in [Docker Hub](https://hub.docker.com/search?q=DolphinScheduler).
|
||||
|
||||
* If you want to modify DolphinScheduler source code, and build Docker images locally, you can run when finished the modification
|
||||
|
||||
```shell
|
||||
cd dolphinscheduler
|
||||
./mvnw -B clean package \
|
||||
|
|
@ -59,6 +61,7 @@ cd dolphinscheduler
|
|||
When the command is finished you could find them by command `docker imaegs`.
|
||||
|
||||
* If you want to modify DolphinScheduler source code, build and push Docker images to your registry <HUB_URL>,you can run when finished the modification
|
||||
|
||||
```shell
|
||||
cd dolphinscheduler
|
||||
./mvnw -B clean deploy \
|
||||
|
|
@ -92,7 +95,6 @@ RUN apt update ; \
|
|||
>
|
||||
> Have to use version after Docker 19.03, because after 19.03 docker contains buildx
|
||||
|
||||
|
||||
## Notice
|
||||
|
||||
There are two ways to configure the DolphinScheduler development environment, standalone mode and normal mode
|
||||
|
|
@ -122,6 +124,7 @@ Find the class `org.apache.dolphinscheduler.StandaloneServer` in Intellij IDEA a
|
|||
### Start frontend server
|
||||
|
||||
Install frontend dependencies and run it.
|
||||
|
||||
> Note: You can see more detail about the frontend setting in [frontend development](./frontend-development.md).
|
||||
|
||||
```shell
|
||||
|
|
@ -130,7 +133,7 @@ pnpm install
|
|||
pnpm run dev
|
||||
```
|
||||
|
||||
The browser access address [http://localhost:3000](http://localhost:3000) can login DolphinScheduler UI. The default username and password are **admin/dolphinscheduler123**
|
||||
The browser access address [http://localhost:5173](http://localhost:5173) can login DolphinScheduler UI. The default username and password are **admin/dolphinscheduler123**
|
||||
|
||||
## DolphinScheduler Normal Mode
|
||||
|
||||
|
|
@ -148,7 +151,6 @@ Download [ZooKeeper](https://www.apache.org/dyn/closer.lua/zookeeper/zookeeper-3
|
|||
dataDir=/data/zookeeper/data
|
||||
dataLogDir=/data/zookeeper/datalog
|
||||
```
|
||||
|
||||
* Run `./bin/zkServer.sh` in terminal by command `./bin/zkServer.sh start`.
|
||||
|
||||
#### Database
|
||||
|
|
@ -166,13 +168,14 @@ Following steps will guide how to start the DolphinScheduler backend service
|
|||
* Open project: Use IDE open the project, here we use Intellij IDEA as an example, after opening it will take a while for Intellij IDEA to complete the dependent download
|
||||
|
||||
* File change
|
||||
|
||||
* If you use MySQL as your metadata database, you need to modify `dolphinscheduler/pom.xml` and change the `scope` of the `mysql-connector-java` dependency to `compile`. This step is not necessary to use PostgreSQL
|
||||
* Modify database configuration, modify the database configuration in the `dolphinscheduler-master/src/main/resources/application.yaml`
|
||||
* Modify database configuration, modify the database configuration in the `dolphinscheduler-worker/src/main/resources/application.yaml`
|
||||
* Modify database configuration, modify the database configuration in the `dolphinscheduler-api/src/main/resources/application.yaml`
|
||||
|
||||
|
||||
We here use MySQL with database, username, password named dolphinscheduler as an example
|
||||
|
||||
```application.yaml
|
||||
spring:
|
||||
datasource:
|
||||
|
|
@ -220,4 +223,4 @@ pnpm install
|
|||
pnpm run dev
|
||||
```
|
||||
|
||||
The browser access address [http://localhost:3000](http://localhost:3000) can login DolphinScheduler UI. The default username and password are **admin/dolphinscheduler123**
|
||||
The browser access address [http://localhost:5173](http://localhost:5173) can login DolphinScheduler UI. The default username and password are **admin/dolphinscheduler123**
|
||||
|
|
|
|||
|
|
@ -1,6 +1,7 @@
|
|||
# Front-end development documentation
|
||||
|
||||
### Technical selection
|
||||
|
||||
```
|
||||
Vue mvvm framework
|
||||
|
||||
|
|
@ -17,10 +18,16 @@ Lodash high performance JavaScript utility library
|
|||
|
||||
### Development environment
|
||||
|
||||
- #### Node installation
|
||||
-
|
||||
|
||||
#### Node installation
|
||||
|
||||
Node package download (note version v12.20.2) `https://nodejs.org/download/release/v12.20.2/`
|
||||
|
||||
- #### Front-end project construction
|
||||
-
|
||||
|
||||
#### Front-end project construction
|
||||
|
||||
Use the command line mode `cd` enter the `dolphinscheduler-ui` project directory and execute `npm install` to pull the project dependency package.
|
||||
|
||||
> If `npm install` is very slow, you can set the taobao mirror
|
||||
|
|
@ -36,13 +43,16 @@ npm config set registry http://registry.npm.taobao.org/
|
|||
API_BASE = http://127.0.0.1:12345
|
||||
```
|
||||
|
||||
> ##### ! ! ! Special attention here. If the project reports a "node-sass error" error while pulling the dependency package, execute the following command again after execution.
|
||||
##### ! ! ! Special attention here. If the project reports a "node-sass error" error while pulling the dependency package, execute the following command again after execution.
|
||||
|
||||
```bash
|
||||
npm install node-sass --unsafe-perm #Install node-sass dependency separately
|
||||
```
|
||||
|
||||
- #### Development environment operation
|
||||
-
|
||||
|
||||
#### Development environment operation
|
||||
|
||||
- `npm start` project development environment (after startup address http://localhost:8888)
|
||||
|
||||
#### Front-end project release
|
||||
|
|
@ -140,6 +150,7 @@ Public module and utill `src/js/module`
|
|||
Home => `http://localhost:8888/#/home`
|
||||
|
||||
Project Management => `http://localhost:8888/#/projects/list`
|
||||
|
||||
```
|
||||
| Project Home
|
||||
| Workflow
|
||||
|
|
@ -149,6 +160,7 @@ Project Management => `http://localhost:8888/#/projects/list`
|
|||
```
|
||||
|
||||
Resource Management => `http://localhost:8888/#/resource/file`
|
||||
|
||||
```
|
||||
| File Management
|
||||
| udf Management
|
||||
|
|
@ -159,6 +171,7 @@ Resource Management => `http://localhost:8888/#/resource/file`
|
|||
Data Source Management => `http://localhost:8888/#/datasource/list`
|
||||
|
||||
Security Center => `http://localhost:8888/#/security/tenant`
|
||||
|
||||
```
|
||||
| Tenant Management
|
||||
| User Management
|
||||
|
|
@ -174,16 +187,19 @@ User Center => `http://localhost:8888/#/user/account`
|
|||
The project `src/js/conf/home` is divided into
|
||||
|
||||
`pages` => route to page directory
|
||||
|
||||
```
|
||||
The page file corresponding to the routing address
|
||||
```
|
||||
|
||||
`router` => route management
|
||||
|
||||
```
|
||||
vue router, the entry file index.js in each page will be registered. Specific operations: https://router.vuejs.org/zh/
|
||||
```
|
||||
|
||||
`store` => status management
|
||||
|
||||
```
|
||||
The page corresponding to each route has a state management file divided into:
|
||||
|
||||
|
|
@ -201,9 +217,13 @@ Specific action:https://vuex.vuejs.org/zh/
|
|||
```
|
||||
|
||||
## specification
|
||||
|
||||
## Vue specification
|
||||
|
||||
##### 1.Component name
|
||||
|
||||
The component is named multiple words and is connected with a wire (-) to avoid conflicts with HTML tags and a clearer structure.
|
||||
|
||||
```
|
||||
// positive example
|
||||
export default {
|
||||
|
|
@ -212,7 +232,9 @@ export default {
|
|||
```
|
||||
|
||||
##### 2.Component files
|
||||
|
||||
The internal common component of the `src/js/module/components` project writes the folder name with the same name as the file name. The subcomponents and util tools that are split inside the common component are placed in the internal `_source` folder of the component.
|
||||
|
||||
```
|
||||
└── components
|
||||
├── header
|
||||
|
|
@ -228,6 +250,7 @@ The internal common component of the `src/js/module/components` project writes t
|
|||
```
|
||||
|
||||
##### 3.Prop
|
||||
|
||||
When you define Prop, you should always name it in camel format (camelCase) and use the connection line (-) when assigning values to the parent component.
|
||||
This follows the characteristics of each language, because it is case-insensitive in HTML tags, and the use of links is more friendly; in JavaScript, the more natural is the hump name.
|
||||
|
||||
|
|
@ -270,7 +293,9 @@ props: {
|
|||
```
|
||||
|
||||
##### 4.v-for
|
||||
|
||||
When performing v-for traversal, you should always bring a key value to make rendering more efficient when updating the DOM.
|
||||
|
||||
```
|
||||
<ul>
|
||||
<li v-for="item in list" :key="item.id">
|
||||
|
|
@ -280,6 +305,7 @@ When performing v-for traversal, you should always bring a key value to make ren
|
|||
```
|
||||
|
||||
v-for should be avoided on the same element as v-if (`for example: <li>`) because v-for has a higher priority than v-if. To avoid invalid calculations and rendering, you should try to use v-if Put it on top of the container's parent element.
|
||||
|
||||
```
|
||||
<ul v-if="showList">
|
||||
<li v-for="item in list" :key="item.id">
|
||||
|
|
@ -289,7 +315,9 @@ v-for should be avoided on the same element as v-if (`for example: <li>`) becaus
|
|||
```
|
||||
|
||||
##### 5.v-if / v-else-if / v-else
|
||||
|
||||
If the elements in the same set of v-if logic control are logically identical, Vue reuses the same part for more efficient element switching, `such as: value`. In order to avoid the unreasonable effect of multiplexing, you should add key to the same element for identification.
|
||||
|
||||
```
|
||||
<div v-if="hasData" key="mazey-data">
|
||||
<span>{{ mazeyData }}</span>
|
||||
|
|
@ -300,12 +328,15 @@ If the elements in the same set of v-if logic control are logically identical, V
|
|||
```
|
||||
|
||||
##### 6.Instruction abbreviation
|
||||
|
||||
In order to unify the specification, the instruction abbreviation is always used. Using `v-bind`, `v-on` is not bad. Here is only a unified specification.
|
||||
|
||||
```
|
||||
<input :value="mazeyUser" @click="verifyUser">
|
||||
```
|
||||
|
||||
##### 7.Top-level element order of single file components
|
||||
|
||||
Styles are packaged in a file, all the styles defined in a single vue file, the same name in other files will also take effect. All will have a top class name before creating a component.
|
||||
Note: The sass plugin has been added to the project, and the sas syntax can be written directly in a single vue file.
|
||||
For uniformity and ease of reading, they should be placed in the order of `<template>`、`<script>`、`<style>`.
|
||||
|
|
@ -357,25 +388,31 @@ For uniformity and ease of reading, they should be placed in the order of `<tem
|
|||
## JavaScript specification
|
||||
|
||||
##### 1.var / let / const
|
||||
|
||||
It is recommended to no longer use var, but use let / const, prefer const. The use of any variable must be declared in advance, except that the function defined by function can be placed anywhere.
|
||||
|
||||
##### 2.quotes
|
||||
|
||||
```
|
||||
const foo = 'after division'
|
||||
const bar = `${foo},ront-end engineer`
|
||||
```
|
||||
|
||||
##### 3.function
|
||||
|
||||
Anonymous functions use the arrow function uniformly. When multiple parameters/return values are used, the object's structure assignment is used first.
|
||||
|
||||
```
|
||||
function getPersonInfo ({name, sex}) {
|
||||
// ...
|
||||
return {name, gender}
|
||||
}
|
||||
```
|
||||
|
||||
The function name is uniformly named with a camel name. The beginning of the capital letter is a constructor. The lowercase letters start with ordinary functions, and the new operator should not be used to operate ordinary functions.
|
||||
|
||||
##### 4.object
|
||||
|
||||
```
|
||||
const foo = {a: 0, b: 1}
|
||||
const bar = JSON.parse(JSON.stringify(foo))
|
||||
|
|
@ -393,7 +430,9 @@ for (let [key, value] of myMap.entries()) {
|
|||
```
|
||||
|
||||
##### 5.module
|
||||
|
||||
Unified management of project modules using import / export.
|
||||
|
||||
```
|
||||
// lib.js
|
||||
export default {}
|
||||
|
|
@ -411,13 +450,16 @@ If the module has only one output value, use `export default`,otherwise no.
|
|||
##### 1.Label
|
||||
|
||||
Do not write the type attribute when referencing external CSS or JavaScript. The HTML5 default type is the text/css and text/javascript properties, so there is no need to specify them.
|
||||
|
||||
```
|
||||
<link rel="stylesheet" href="//www.test.com/css/test.css">
|
||||
<script src="//www.test.com/js/test.js"></script>
|
||||
```
|
||||
|
||||
##### 2.Naming
|
||||
|
||||
The naming of Class and ID should be semantic, and you can see what you are doing by looking at the name; multiple words are connected by a link.
|
||||
|
||||
```
|
||||
// positive example
|
||||
.test-header{
|
||||
|
|
@ -426,6 +468,7 @@ The naming of Class and ID should be semantic, and you can see what you are doin
|
|||
```
|
||||
|
||||
##### 3.Attribute abbreviation
|
||||
|
||||
CSS attributes use abbreviations as much as possible to improve the efficiency and ease of understanding of the code.
|
||||
|
||||
```
|
||||
|
|
@ -439,6 +482,7 @@ border: 1px solid #ccc;
|
|||
```
|
||||
|
||||
##### 4.Document type
|
||||
|
||||
The HTML5 standard should always be used.
|
||||
|
||||
```
|
||||
|
|
@ -446,7 +490,9 @@ The HTML5 standard should always be used.
|
|||
```
|
||||
|
||||
##### 5.Notes
|
||||
|
||||
A block comment should be written to a module file.
|
||||
|
||||
```
|
||||
/**
|
||||
* @module mazey/api
|
||||
|
|
@ -458,6 +504,7 @@ A block comment should be written to a module file.
|
|||
## interface
|
||||
|
||||
##### All interfaces are returned as Promise
|
||||
|
||||
Note that non-zero is wrong for catching catch
|
||||
|
||||
```
|
||||
|
|
@ -477,6 +524,7 @@ test.then(res => {
|
|||
```
|
||||
|
||||
Normal return
|
||||
|
||||
```
|
||||
{
|
||||
code:0,
|
||||
|
|
@ -486,6 +534,7 @@ Normal return
|
|||
```
|
||||
|
||||
Error return
|
||||
|
||||
```
|
||||
{
|
||||
code:10000,
|
||||
|
|
@ -493,8 +542,10 @@ Error return
|
|||
msg:'failed'
|
||||
}
|
||||
```
|
||||
|
||||
If the interface is a post request, the Content-Type defaults to application/x-www-form-urlencoded; if the Content-Type is changed to application/json,
|
||||
Interface parameter transfer needs to be changed to the following way
|
||||
|
||||
```
|
||||
io.post('url', payload, null, null, { emulateJSON: false } res => {
|
||||
resolve(res)
|
||||
|
|
@ -524,6 +575,7 @@ User Center Related Interfaces `src/js/conf/home/store/user/actions.js`
|
|||
(1) First place the icon icon of the node in the `src/js/conf/home/pages/dag/img `folder, and note the English name of the node defined by the `toolbar_${in the background. For example: SHELL}.png`
|
||||
|
||||
(2) Find the `tasksType` object in `src/js/conf/home/pages/dag/_source/config.js` and add it to it.
|
||||
|
||||
```
|
||||
'DEPENDENT': { // The background definition node type English name is used as the key value
|
||||
desc: 'DEPENDENT', // tooltip desc
|
||||
|
|
@ -532,6 +584,7 @@ User Center Related Interfaces `src/js/conf/home/store/user/actions.js`
|
|||
```
|
||||
|
||||
(3) Add a `${node type (lowercase)}`.vue file in `src/js/conf/home/pages/dag/_source/formModel/tasks`. The contents of the components related to the current node are written here. Must belong to a node component must have a function _verification () After the verification is successful, the relevant data of the current component is thrown to the parent component.
|
||||
|
||||
```
|
||||
/**
|
||||
* Verification
|
||||
|
|
@ -566,6 +619,7 @@ User Center Related Interfaces `src/js/conf/home/store/user/actions.js`
|
|||
(4) Common components used inside the node component are under` _source`, and `commcon.js` is used to configure public data.
|
||||
|
||||
##### 2.Increase the status type
|
||||
|
||||
(1) Find the `tasksState` object in `src/js/conf/home/pages/dag/_source/config.js` and add it to it.
|
||||
|
||||
```
|
||||
|
|
@ -579,7 +633,9 @@ User Center Related Interfaces `src/js/conf/home/store/user/actions.js`
|
|||
```
|
||||
|
||||
##### 3.Add the action bar tool
|
||||
|
||||
(1) Find the `toolOper` object in `src/js/conf/home/pages/dag/_source/config.js` and add it to it.
|
||||
|
||||
```
|
||||
{
|
||||
code: 'pointer', // tool identifier
|
||||
|
|
@ -599,13 +655,12 @@ User Center Related Interfaces `src/js/conf/home/store/user/actions.js`
|
|||
|
||||
`util.js` => belongs to the `plugIn` tool class
|
||||
|
||||
|
||||
The operation is handled in the `src/js/conf/home/pages/dag/_source/dag.js` => `toolbarEvent` event.
|
||||
|
||||
|
||||
##### 3.Add a routing page
|
||||
|
||||
(1) First add a routing address`src/js/conf/home/router/index.js` in route management
|
||||
|
||||
```
|
||||
routing address{
|
||||
path: '/test', // routing address
|
||||
|
|
@ -621,10 +676,10 @@ routing address{
|
|||
|
||||
This will give you direct access to`http://localhost:8888/#/test`
|
||||
|
||||
|
||||
##### 4.Increase the preset mailbox
|
||||
|
||||
Find the `src/lib/localData/email.js` startup and timed email address input to automatically pull down the match.
|
||||
|
||||
```
|
||||
export default ["test@analysys.com.cn","test1@analysys.com.cn","test3@analysys.com.cn"]
|
||||
```
|
||||
|
|
|
|||
|
|
@ -21,8 +21,9 @@ Some quick tips when using email:
|
|||
- Tagging the subject line of your email will help you get a faster response, e.g. [api-server]: How to get open api interface?
|
||||
|
||||
- Tags may help identify a topic by:
|
||||
|
||||
- Component: MasterServer,ApiServer,WorkerServer,AlertServer, etc
|
||||
- Level: Beginner, Intermediate, Advanced
|
||||
- Scenario: Debug, How-to
|
||||
|
||||
- For error logs or long code examples, please use [GitHub gist](https://gist.github.com/) and include only a few lines of the pertinent code / log within the email.
|
||||
|
||||
|
|
|
|||
|
|
@ -20,7 +20,6 @@ Moreover, when we intend to refer a new software ( not limited to 3rd party jar,
|
|||
|
||||
* [COMMUNITY-LED DEVELOPMENT "THE APACHE WAY"](https://apache.org/dev/licensing-howto.html)
|
||||
|
||||
|
||||
For example, we should contain the NOTICE file (every open-source project has NOTICE file, generally under root directory) of ZooKeeper in our project when we are using ZooKeeper. As the Apache explains, "Work" shall mean the work of authorship, whether in Source or Object form, made available under the License, as indicated by a copyright notice that is included in or attached to the work.
|
||||
|
||||
We are not going to dive into every 3rd party open-source license policy, you may look up them if interested.
|
||||
|
|
@ -40,3 +39,4 @@ We need to follow the following steps when we need to add new jars or external r
|
|||
|
||||
* [COMMUNITY-LED DEVELOPMENT "THE APACHE WAY"](https://apache.org/dev/licensing-howto.html)
|
||||
* [ASF 3RD PARTY LICENSE POLICY](https://apache.org/legal/resolved.html)
|
||||
|
||||
|
|
|
|||
|
|
@ -3,6 +3,7 @@
|
|||
The following Code of Conduct is based on full compliance with the [Apache Software Foundation Code of Conduct](https://www.apache.org/foundation/policies/conduct.html).
|
||||
|
||||
## Development philosophy
|
||||
|
||||
- **Consistent** code style, naming, and usage are consistent.
|
||||
- **Easy to read** code is obvious, easy to read and understand, when debugging one knows the intent of the code.
|
||||
- **Neat** agree with the concepts of《Refactoring》and《Code Cleanliness》and pursue clean and elegant code.
|
||||
|
|
@ -63,6 +64,6 @@ The following Code of Conduct is based on full compliance with the [Apache Softw
|
|||
- Accurate assertion, try not to use `not`,`containsString` assertion.
|
||||
- The true value of the test case should be named actualXXX, and the expected value should be named expectedXXX.
|
||||
- Classes and Methods with `@Test` labels do not require javadoc.
|
||||
|
||||
- Public specifications.
|
||||
- Each line is no longer than `200` in length, ensuring that each line is semantically complete for easy understanding.
|
||||
|
||||
|
|
|
|||
|
|
@ -1,6 +1,7 @@
|
|||
# Issue Notice
|
||||
|
||||
## Preface
|
||||
|
||||
Issues function is used to track various Features, Bugs, Functions, etc. The project maintainer can organize the tasks to be completed through issues.
|
||||
|
||||
Issue is an important step in drawing out a feature or bug,
|
||||
|
|
|
|||
|
|
@ -1,6 +1,7 @@
|
|||
# Pull Request Notice
|
||||
|
||||
## Preface
|
||||
|
||||
Pull Request is a way of software cooperation, which is a process of bringing code involving different functions into the trunk. During this process, the code can be discussed, reviewed, and modified.
|
||||
|
||||
In Pull Request, we try not to discuss the implementation of the code. The general implementation of the code and its logic should be determined in Issue. In the Pull Request, we only focus on the code format and code specification, so as to avoid wasting time caused by different opinions on implementation.
|
||||
|
|
@ -75,3 +76,4 @@ see [Code Style](../development-environment-setup.md#code-style) for details.
|
|||
the second is multiple issues have subtle differences.
|
||||
In this scenario, the responsibilities of each issue can be clearly divided. The type of each issue is marked as Sub-Task, and then these sub task type issues are associated with one issue.
|
||||
And each Pull Request is submitted should be associated with only one issue of a sub task.
|
||||
|
||||
|
|
|
|||
|
|
@ -28,7 +28,7 @@ go to section [review Pull Requests](#pull-requests).
|
|||
Review Issues means discuss [Issues][all-issues] in GitHub and give suggestions on it. Include but are not limited to the following situations
|
||||
|
||||
| Situation | Reason | Label | Action |
|
||||
| ------ | ------ | ------ | ------ |
|
||||
|-------------------------|-------------------------------|------------------------------------------------------|---------------------------------------------------------------------|
|
||||
| wont fix | Has been fixed in dev branch | [wontfix][label-wontfix] | Close Issue, inform creator the fixed version if it already release |
|
||||
| duplicate issue | Had the same problem before | [duplicate][label-duplicate] | Close issue, inform creator the link of same issue |
|
||||
| Description not clearly | Without detail reproduce step | [need more information][label-need-more-information] | Inform creator add more description |
|
||||
|
|
@ -37,7 +37,7 @@ In addition give suggestion, add label for issue is also important during review
|
|||
better, which convenient for further processing. An issue can with more than one label. Common issue categories are:
|
||||
|
||||
| Label | Meaning |
|
||||
| ------ | ------ |
|
||||
|------------------------------------------|--------------------------------|
|
||||
| [UI][label-UI] | UI and front-end related |
|
||||
| [security][label-security] | Security Issue |
|
||||
| [user experience][label-user-experience] | User experience Issue |
|
||||
|
|
@ -55,7 +55,7 @@ Beside classification, label could also set the priority of Issues. The higher t
|
|||
in the community, the easier it is to be fixed or implemented. The priority label are as follows
|
||||
|
||||
| Label | priority |
|
||||
| ------ | ------ |
|
||||
|------------------------------------------|-----------------|
|
||||
| [priority:high][label-priority-high] | High priority |
|
||||
| [priority:middle][label-priority-middle] | Middle priority |
|
||||
| [priority:low][label-priority-low] | Low priority |
|
||||
|
|
@ -75,7 +75,7 @@ Before reading following content, please make sure you have labeled the Issue.
|
|||
When an Issue need to create Pull Requests, you could also labeled it from below.
|
||||
|
||||
| Label | Mean |
|
||||
| ------ | ------ |
|
||||
|--------------------------------------------|---------------------------------------------|
|
||||
| [Chore][label-Chore] | Chore for project |
|
||||
| [Good first issue][label-good-first-issue] | Good first issue for new contributor |
|
||||
| [easy to fix][label-easy-to-fix] | Easy to fix, harder than `Good first issue` |
|
||||
|
|
@ -90,14 +90,14 @@ When an Issue need to create Pull Requests, you could also labeled it from below
|
|||
<!-- markdown-link-check-disable -->
|
||||
Review Pull mean discussing in [Pull Requests][all-PRs] in GitHub and giving suggestions to it. DolphinScheduler's
|
||||
Pull Requests reviewing are the same as [GitHub's reviewing changes in pull requests][gh-review-pr]. You can give your
|
||||
suggestions in Pull Requests
|
||||
|
||||
suggestions in Pull Reque-->
|
||||
* When you think the Pull Request is OK to be merged, you can agree to the Pull Request according to the "Approve" process
|
||||
in [GitHub's reviewing changes in pull requests][gh-review-pr].
|
||||
* When you think Pull Request needs to be changed, you can comment it according to the "Comment" process in
|
||||
[GitHub's reviewing changes in pull requests][gh-review-pr]. And when you think issues that must be fixed before they
|
||||
merged, please follow "Request changes" in [GitHub's reviewing changes in pull requests][gh-review-pr] to ask contributors
|
||||
modify it.
|
||||
|
||||
<!-- markdown-link-check-enable -->
|
||||
|
||||
Labeled Pull Requests is an important part. Reasonable classification can save a lot of time for reviewers. The good news
|
||||
|
|
@ -108,7 +108,7 @@ and [priority:high][label-priority-high].
|
|||
Pull Requests have some unique labels of it own
|
||||
|
||||
| Label | Mean |
|
||||
| ------ | ------ |
|
||||
|--------------------------------------------------------|----------------------------------------------------------|
|
||||
| [miss document][label-miss-document] | Pull Requests miss document, and should be add |
|
||||
| [first time contributor][label-first-time-contributor] | Pull Requests submit by first time contributor |
|
||||
| [don't merge][label-do-not-merge] | Pull Requests have some problem and should not be merged |
|
||||
|
|
@ -151,3 +151,4 @@ Pull Requests have some unique labels of it own
|
|||
[all-issues]: https://github.com/apache/dolphinscheduler/issues
|
||||
[all-PRs]: https://github.com/apache/dolphinscheduler/pulls
|
||||
[gh-review-pr]: https://docs.github.com/en/pull-requests/collaborating-with-pull-requests/reviewing-changes-in-pull-requests/about-pull-request-reviews
|
||||
|
||||
|
|
|
|||
|
|
@ -3,19 +3,16 @@
|
|||
* First from the remote repository *https://github.com/apache/dolphinscheduler.git* fork a copy of the code into your own repository
|
||||
|
||||
* There are currently three branches in the remote repository:
|
||||
|
||||
* master normal delivery branch
|
||||
After the stable release, merge the code from the stable branch into the master.
|
||||
|
||||
* dev daily development branch
|
||||
Every day dev development branch, newly submitted code can pull request to this branch.
|
||||
|
||||
|
||||
* Clone your repository to your local
|
||||
`git clone https://github.com/apache/dolphinscheduler.git`
|
||||
|
||||
* Add remote repository address, named upstream
|
||||
`git remote add upstream https://github.com/apache/dolphinscheduler.git`
|
||||
|
||||
* View repository
|
||||
`git remote -v`
|
||||
|
||||
|
|
@ -39,6 +36,7 @@ git push --set-upstream origin dev-1.0
|
|||
```
|
||||
|
||||
* Create new branch
|
||||
|
||||
```
|
||||
git checkout -b xxx origin/dev
|
||||
```
|
||||
|
|
@ -60,4 +58,3 @@ Make sure that the branch `xxx` is building successfully on the latest code of t
|
|||
|
||||
* Finally, congratulations, you have become an official contributor to dolphinscheduler!
|
||||
|
||||
|
||||
|
|
|
|||
|
|
@ -21,3 +21,4 @@ Unsubscribe from the mailing list steps are as follows:
|
|||
2. Receive confirmation email and reply. After completing step 1, you will receive a confirmation email from dev-help@dolphinscheduler.apache.org (if not received, please confirm whether the email is automatically classified as spam, promotion email, subscription email, etc.) . Then reply directly to the email, or click on the link in the email to reply quickly, the subject and content are arbitrary.
|
||||
|
||||
3. Receive a goodbye email. After completing the above steps, you will receive a goodbye email with the subject GOODBYE from dev@dolphinscheduler.apache.org, and you have successfully unsubscribed to the Apache DolphinScheduler mailing list, and you will not receive emails from dev@dolphinscheduler.apache.org.
|
||||
|
||||
|
|
|
|||
|
|
@ -12,8 +12,10 @@
|
|||
- Pay attention to boundary conditions.
|
||||
- Unit tests should be well designed as well as avoiding useless code.
|
||||
- When you find a `method` is difficult to write unit test, and if you confirm that the `method` is `bad code`, then refactor it with the developer.
|
||||
|
||||
<!-- markdown-link-check-disable -->
|
||||
- DolphinScheduler: [mockito](http://site.mockito.org/). Here are some development guides: [mockito tutorial](http://www.baeldung.com/bdd-mockito), [mockito refcard](https://dzone.com/refcardz/mockito)
|
||||
|
||||
<!-- markdown-link-check-enable -->
|
||||
- TDD(option): When you start writing a new feature, you can try writing test cases first.
|
||||
|
||||
|
|
@ -100,6 +102,7 @@ The test will fail when the code in the unit test throws an exception. Therefore
|
|||
}
|
||||
}
|
||||
```
|
||||
|
||||
You should this:
|
||||
|
||||
```java
|
||||
|
|
|
|||
|
|
@ -50,3 +50,4 @@ That is, the workflow instance ID and task instance ID are injected in the print
|
|||
- Branch printing of logs is prohibited. The contents of the logs need to be associated with the relevant information in the log format, and printing them in separate lines will cause the contents of the logs to not match the time and other information, and cause the logs to be mixed in a large number of log environments, which will make log retrieval more difficult.
|
||||
- The use of the "+" operator for splicing log content is prohibited. Use placeholders for formatting logs for printing to improve memory usage efficiency.
|
||||
- When the log content includes object instances, you need to make sure to override the toString() method to prevent printing meaningless hashcode.
|
||||
|
||||
|
|
|
|||
|
|
@ -22,7 +22,7 @@ We could reuse the main command the CI run and publish our Docker images to Dock
|
|||
|
||||
## Publish pydolphinscheduler to PyPI
|
||||
|
||||
Python API need to release to PyPI for easier download and use, you can see more detail in [Python API release](https://github.com/apache/dolphinscheduler/blob/dev/dolphinscheduler-python/pydolphinscheduler/RELEASE.md#to-pypi)
|
||||
Python API need to release to PyPI for easier download and use, you can see more detail in [Python API release](https://github.com/apache/dolphinscheduler/blob/3.1.1/dolphinscheduler-python/pydolphinscheduler/RELEASE.md#to-pypi)
|
||||
to finish PyPI release.
|
||||
|
||||
## Get All Contributors
|
||||
|
|
|
|||
|
|
@ -29,3 +29,4 @@ For example, to release `x.y.z`, the following updates are required:
|
|||
- Add new history version
|
||||
- `docs/docs/en/history-versions.md` and `docs/docs/zh/history-versions.md`: Add the new version and link for `x.y.z`
|
||||
- `docs/configs/docsdev.js`: change `/dev/` to `/x.y.z/`
|
||||
|
||||
|
|
|
|||
|
|
@ -210,7 +210,7 @@ git push origin --tags
|
|||
> Note1: In this step, you should use github token for password because native password no longer supported, you can see
|
||||
> https://docs.github.com/en/authentication/keeping-your-account-and-data-secure/creating-a-personal-access-token for more
|
||||
> detail about how to create token about it.
|
||||
|
||||
>
|
||||
> Note2: After the command done, it will auto-created `release.properties` file and `*.Backup` files, their will be need
|
||||
> in the following command and DO NOT DELETE THEM
|
||||
|
||||
|
|
@ -293,6 +293,7 @@ cd ~/ds_svn/dev/dolphinscheduler
|
|||
svn add *
|
||||
svn --username="${A_USERNAME}" commit -m "release ${VERSION}"
|
||||
```
|
||||
|
||||
## Check Release
|
||||
|
||||
### Check sha512 hash
|
||||
|
|
@ -538,3 +539,4 @@ DolphinScheduler Resources:
|
|||
- Mailing list: dev@dolphinscheduler.apache.org
|
||||
- Documents: https://dolphinscheduler.apache.org/zh-cn/docs/<VERSION>/user_doc/about/introduction.html
|
||||
```
|
||||
|
||||
|
|
|
|||
|
|
@ -280,7 +280,7 @@ A : Will hive pom
|
|||
<dependency>
|
||||
<groupId>org.apache.hive</groupId>
|
||||
<artifactId>hive-jdbc</artifactId>
|
||||
<version>2.1.0</version>
|
||||
<version>2.3.9</version>
|
||||
</dependency>
|
||||
```
|
||||
|
||||
|
|
|
|||
|
|
@ -9,7 +9,7 @@ The following shows the `DingTalk` configuration example:
|
|||
## Parameter Configuration
|
||||
|
||||
| **Parameter** | **Description** |
|
||||
| --- | --- |
|
||||
|----------------|-----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|
|
||||
| Warning Type | Alert on success or failure or both. |
|
||||
| WebHook | The format is: [https://oapi.dingtalk.com/robot/send?access\_token=XXXXXX](https://oapi.dingtalk.com/robot/send?access_token=XXXXXX) |
|
||||
| Keyword | Custom keywords for security settings. |
|
||||
|
|
@ -25,3 +25,4 @@ The following shows the `DingTalk` configuration example:
|
|||
## Reference
|
||||
|
||||
- [DingTalk Custom Robot Access Development Documentation](https://open.dingtalk.com/document/robots/custom-robot-access)
|
||||
|
||||
|
|
|
|||
|
|
@ -1,4 +1,5 @@
|
|||
# Email
|
||||
|
||||
If you need to use `Email` for alerting, create an alert instance in the alert instance management and select the Email plugin.
|
||||
|
||||
The following shows the `Email` configuration example:
|
||||
|
|
|
|||
|
|
@ -8,7 +8,7 @@ The following is the `WebexTeams` configuration example:
|
|||
## Parameter Configuration
|
||||
|
||||
| **Parameter** | **Description** |
|
||||
| --- | --- |
|
||||
|-----------------|-------------------------------------------------------------------------------------------------------------------------|
|
||||
| botAccessToken | The access token of robot. |
|
||||
| roomID | The ID of the room that receives message (only support one room ID). |
|
||||
| toPersonId | The person ID of the recipient when sending a private 1:1 message. |
|
||||
|
|
@ -59,3 +59,4 @@ The `Room ID` we can acquire it from the `id` of creating a new group chat room
|
|||
|
||||
- [WebexTeams Application Bot Guide](https://developer.webex.com/docs/bots)
|
||||
- [WebexTeams Message Guide](https://developer.webex.com/docs/api/v1/messages/create-a-message)
|
||||
|
||||
|
|
|
|||
|
|
@ -40,7 +40,6 @@ The following is the `query userId` API example:
|
|||
|
||||
APP: https://work.weixin.qq.com/api/doc/90000/90135/90236
|
||||
|
||||
|
||||
### Group Chat
|
||||
|
||||
The Group Chat send type means to notify the alert results via group chat created by Enterprise WeChat API, sending messages to all members of the group and specified users are not supported.
|
||||
|
|
@ -69,3 +68,4 @@ The following is the `create new group chat` API and `query userId` API example:
|
|||
## Reference
|
||||
|
||||
- Group Chat:https://work.weixin.qq.com/api/doc/90000/90135/90248
|
||||
|
||||
|
|
|
|||
|
|
@ -10,6 +10,7 @@ The following shows the `Feishu` configuration example:
|
|||
## Parameter Configuration
|
||||
|
||||
* Webhook
|
||||
|
||||
> Copy the robot webhook URL shown below:
|
||||
|
||||

|
||||
|
|
|
|||
|
|
@ -5,7 +5,7 @@ If you need to use `Http script` for alerting, create an alert instance in the a
|
|||
## Parameter Configuration
|
||||
|
||||
| **Parameter** | **Description** |
|
||||
| --- | --- |
|
||||
|---------------|-----------------------------------------------------------------------------------------------------|
|
||||
| URL | The `Http` request URL needs to contain protocol, host, path and parameters if the method is `GET`. |
|
||||
| Request Type | Select the request type from `POST` or `GET`. |
|
||||
| Headers | The headers of the `Http` request in JSON format. |
|
||||
|
|
|
|||
|
|
@ -8,7 +8,7 @@ The following shows the `Script` configuration example:
|
|||
## Parameter Configuration
|
||||
|
||||
| **Parameter** | **Description** |
|
||||
| --- | --- |
|
||||
|---------------|--------------------------------------------------|
|
||||
| User Params | User defined parameters will pass to the script. |
|
||||
| Script Path | The file location path in the server. |
|
||||
| Type | Support `Shell` script. |
|
||||
|
|
|
|||
|
|
@ -8,7 +8,7 @@ The following shows the `Telegram` configuration example:
|
|||
## Parameter Configuration
|
||||
|
||||
| **Parameter** | **Description** |
|
||||
| --- | --- |
|
||||
|---------------|---------------------------------------------------------------|
|
||||
| WebHook | The WebHook of Telegram when use robot to send message. |
|
||||
| botToken | The access token of robot. |
|
||||
| chatId | Sub Telegram Channel. |
|
||||
|
|
@ -35,3 +35,4 @@ The webhook needs to be able to receive and use the same JSON body of HTTP POST
|
|||
- [Telegram Application Bot Guide](https://core.telegram.org/bots)
|
||||
- [Telegram Bots Api](https://core.telegram.org/bots/api)
|
||||
- [Telegram SendMessage Api](https://core.telegram.org/bots/api#sendmessage)
|
||||
|
||||
|
|
|
|||
|
|
@ -1,4 +1,5 @@
|
|||
# Data Quality
|
||||
|
||||
## Introduction
|
||||
|
||||
The data quality task is used to check the data accuracy during the integration and processing of data. Data quality tasks in this release include single-table checking, single-table custom SQL checking, multi-table accuracy, and two-table value comparisons. The running environment of the data quality task is Spark 2.4.0, and other versions have not been verified, and users can verify by themselves.
|
||||
|
|
@ -28,12 +29,12 @@ data-quality.jar.name=dolphinscheduler-data-quality-dev-SNAPSHOT.jar
|
|||
## Detailed Inspection Logic
|
||||
|
||||
| **Parameter** | **Description** |
|
||||
| ----- | ---- |
|
||||
|---------------|----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|
|
||||
| CheckMethod | [CheckFormula][Operator][Threshold], if the result is true, it indicates that the data does not meet expectations, and the failure strategy is executed. |
|
||||
| CheckFormula | <ul><li>Expected-Actual</li><li>Actual-Expected</li><li>(Actual/Expected)x100%</li><li>(Expected-Actual)/Expected x100%</li></ul> |
|
||||
| Operator | =, >, >=, <, <=, != |
|
||||
| ExpectedValue | <ul><li>FixValue</li><li>DailyAvg</li><li>WeeklyAvg</li><li>MonthlyAvg</li><li>Last7DayAvg</li><li>Last30DayAvg</li><li>SrcTableTotalRows</li><li>TargetTableTotalRows</li></ul> |
|
||||
| Example |<ul><li>CheckFormula:Expected-Actual</li><li>Operator:></li><li>Threshold:0</li><li>ExpectedValue:FixValue=9</li></ul>
|
||||
| Example | <ul><li>CheckFormula:Expected-Actual</li><li>Operator:></li><li>Threshold:0</li><li>ExpectedValue:FixValue=9</li></ul> |
|
||||
|
||||
In the example, assuming that the actual value is 10, the operator is >, and the expected value is 9, then the result 10 -9 > 0 is true, which means that the row data in the empty column has exceeded the threshold, and the task is judged to fail.
|
||||
|
||||
|
|
@ -50,7 +51,6 @@ The goal of the null value check is to check the number of empty rows in the spe
|
|||
```sql
|
||||
SELECT COUNT(*) AS miss FROM ${src_table} WHERE (${src_field} is null or ${src_field} = '') AND (${src_filter})
|
||||
```
|
||||
|
||||
- The SQL to calculate the total number of rows in the table is as follows:
|
||||
|
||||
```sql
|
||||
|
|
@ -62,7 +62,7 @@ The goal of the null value check is to check the number of empty rows in the spe
|
|||

|
||||
|
||||
| **Parameter** | **Description** |
|
||||
| ----- | ---- |
|
||||
|------------------------|-----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|
|
||||
| Source data type | Select MySQL, PostgreSQL, etc. |
|
||||
| Source data source | The corresponding data source under the source data type. |
|
||||
| Source data table | Drop-down to select the table where the validation data is located. |
|
||||
|
|
@ -75,7 +75,9 @@ The goal of the null value check is to check the number of empty rows in the spe
|
|||
| Expected value type | Select the desired type from the drop-down menu. |
|
||||
|
||||
## Timeliness Check of Single Table Check
|
||||
|
||||
### Inspection Introduction
|
||||
|
||||
The timeliness check is used to check whether the data is processed within the expected time. The start time and end time can be specified to define the time range. If the amount of data within the time range does not reach the set threshold, the check task will be judged as fail.
|
||||
|
||||
### Interface Operation Guide
|
||||
|
|
@ -83,9 +85,9 @@ The timeliness check is used to check whether the data is processed within the e
|
|||

|
||||
|
||||
| **Parameter** | **Description** |
|
||||
| ----- | ---- |
|
||||
| Source data type | Select MySQL, PostgreSQL, etc.
|
||||
| Source data source | The corresponding data source under the source data type.
|
||||
|------------------------|-----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|
|
||||
| Source data type | Select MySQL, PostgreSQL, etc. |
|
||||
| Source data source | The corresponding data source under the source data type. |
|
||||
| Source data table | Drop-down to select the table where the validation data is located. |
|
||||
| Src filter conditions | Such as the title, it will also be used when counting the total number of rows in the table, optional. |
|
||||
| Src table check column | Drop-down to select check column name. |
|
||||
|
|
@ -101,6 +103,7 @@ The timeliness check is used to check whether the data is processed within the e
|
|||
## Field Length Check for Single Table Check
|
||||
|
||||
### Inspection Introduction
|
||||
|
||||
The goal of field length verification is to check whether the length of the selected field meets the expectations. If there is data that does not meet the requirements, and the number of rows exceeds the threshold, the task will be judged to fail.
|
||||
|
||||
### Interface Operation Guide
|
||||
|
|
@ -108,7 +111,7 @@ The goal of field length verification is to check whether the length of the sele
|
|||

|
||||
|
||||
| **Parameter** | **Description** |
|
||||
| ----- | ---- |
|
||||
|------------------------|-----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|
|
||||
| Source data type | Select MySQL, PostgreSQL, etc. |
|
||||
| Source data source | The corresponding data source under the source data type. |
|
||||
| Source data table | Drop-down to select the table where the validation data is located. |
|
||||
|
|
@ -125,6 +128,7 @@ The goal of field length verification is to check whether the length of the sele
|
|||
## Uniqueness Check for Single Table Check
|
||||
|
||||
### Inspection Introduction
|
||||
|
||||
The goal of the uniqueness check is to check whether the fields are duplicated. It is generally used to check whether the primary key is duplicated. If there are duplicates and the threshold is reached, the check task will be judged to be failed.
|
||||
|
||||
### Interface Operation Guide
|
||||
|
|
@ -132,7 +136,7 @@ The goal of the uniqueness check is to check whether the fields are duplicated.
|
|||

|
||||
|
||||
| **Parameter** | **Description** |
|
||||
| ----- | ---- |
|
||||
|------------------------|-----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|
|
||||
| Source data type | Select MySQL, PostgreSQL, etc. |
|
||||
| Source data source | The corresponding data source under the source data type. |
|
||||
| Source data table | Drop-down to select the table where the validation data is located. |
|
||||
|
|
@ -147,6 +151,7 @@ The goal of the uniqueness check is to check whether the fields are duplicated.
|
|||
## Regular Expression Check for Single Table Check
|
||||
|
||||
### Inspection Introduction
|
||||
|
||||
The goal of regular expression verification is to check whether the format of the value of a field meets the requirements, such as time format, email format, ID card format, etc. If there is data that does not meet the format and exceeds the threshold, the task will be judged as failed.
|
||||
|
||||
### Interface Operation Guide
|
||||
|
|
@ -154,7 +159,7 @@ The goal of regular expression verification is to check whether the format of th
|
|||

|
||||
|
||||
| **Parameter** | **Description** |
|
||||
| ----- | ---- |
|
||||
|------------------------|-----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|
|
||||
| Source data type | Select MySQL, PostgreSQL, etc. |
|
||||
| Source data source | The corresponding data source under the source data type. |
|
||||
| Source data table | Drop-down to select the table where the validation data is located. |
|
||||
|
|
@ -168,7 +173,9 @@ The goal of regular expression verification is to check whether the format of th
|
|||
| Expected value type | Select the desired type from the drop-down menu. |
|
||||
|
||||
## Enumeration Value Validation for Single Table Check
|
||||
|
||||
### Inspection Introduction
|
||||
|
||||
The goal of enumeration value verification is to check whether the value of a field is within the range of the enumeration value. If there is data that is not in the range of the enumeration value and exceeds the threshold, the task will be judged to fail.
|
||||
|
||||
### Interface Operation Guide
|
||||
|
|
@ -176,7 +183,7 @@ The goal of enumeration value verification is to check whether the value of a fi
|
|||

|
||||
|
||||
| **Parameter** | **Description** |
|
||||
| ----- | ---- |
|
||||
|-----------------------------|-----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|
|
||||
| Source data type | Select MySQL, PostgreSQL, etc. |
|
||||
| Source data source | The corresponding data source under the source data type. |
|
||||
| Source data table | Drop-down to select the table where the validation data is located. |
|
||||
|
|
@ -192,6 +199,7 @@ The goal of enumeration value verification is to check whether the value of a fi
|
|||
## Table Row Number Verification for Single Table Check
|
||||
|
||||
### Inspection Introduction
|
||||
|
||||
The goal of table row number verification is to check whether the number of rows in the table reaches the expected value. If the number of rows does not meet the standard, the task will be judged as failed.
|
||||
|
||||
### Interface Operation Guide
|
||||
|
|
@ -199,7 +207,7 @@ The goal of table row number verification is to check whether the number of rows
|
|||

|
||||
|
||||
| **Parameter** | **Description** |
|
||||
| ----- | ---- |
|
||||
|------------------------|-----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|
|
||||
| Source data type | Select MySQL, PostgreSQL, etc. |
|
||||
| Source data source | The corresponding data source under the source data type. |
|
||||
| Source data table | Drop-down to select the table where the validation data is located. |
|
||||
|
|
@ -218,7 +226,7 @@ The goal of table row number verification is to check whether the number of rows
|
|||

|
||||
|
||||
| **Parameter** | **Description** |
|
||||
| ----- | ---- |
|
||||
|------------------------------|--------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|
|
||||
| Source data type | Select MySQL, PostgreSQL, etc. |
|
||||
| Source data source | The corresponding data source under the source data type. |
|
||||
| Source data table | Drop-down to select the table where the data to be verified is located. |
|
||||
|
|
@ -232,12 +240,14 @@ The goal of table row number verification is to check whether the number of rows
|
|||
| Expected value type | Select the desired type from the drop-down menu. |
|
||||
|
||||
## Accuracy Check of Multi-table
|
||||
|
||||
### Inspection Introduction
|
||||
|
||||
Accuracy checks are performed by comparing the accuracy differences of data records for selected fields between two tables, examples are as follows
|
||||
- table test1
|
||||
|
||||
| c1 | c2 |
|
||||
| :---: | :---: |
|
||||
|:--:|:--:|
|
||||
| a | 1 |
|
||||
| b | 2 |
|
||||
|
||||
|
|
@ -255,7 +265,7 @@ If you compare the data in c1 and c21, the tables test1 and test2 are exactly th
|
|||

|
||||
|
||||
| **Parameter** | **Description** |
|
||||
| ----- | ---- |
|
||||
|--------------------------|----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|
|
||||
| Source data type | Select MySQL, PostgreSQL, etc. |
|
||||
| Source data source | The corresponding data source under the source data type. |
|
||||
| Source data table | Drop-down to select the table where the data to be verified is located. |
|
||||
|
|
@ -271,7 +281,9 @@ If you compare the data in c1 and c21, the tables test1 and test2 are exactly th
|
|||
| Expected value type | Select the desired type in the drop-down menu, only `SrcTableTotalRow`, `TargetTableTotalRow` and fixed value are suitable for selection here. |
|
||||
|
||||
## Comparison of the values checked by the two tables
|
||||
|
||||
### Inspection Introduction
|
||||
|
||||
Two-table value comparison allows users to customize different SQL statistics for two tables and compare the corresponding values. For example, for the source table A, the total amount of a certain column is calculated, and for the target table, the total amount of a certain column is calculated. value sum2, compare sum1 and sum2 to determine the check result.
|
||||
|
||||
### Interface Operation Guide
|
||||
|
|
@ -279,7 +291,7 @@ Two-table value comparison allows users to customize different SQL statistics fo
|
|||

|
||||
|
||||
| **Parameter** | **Description** |
|
||||
| ----- | ---- |
|
||||
|--------------------------------|---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|
|
||||
| Source data type | Select MySQL, PostgreSQL, etc. |
|
||||
| Source data source | The corresponding data source under the source data type. |
|
||||
| Source data table | The table where the data is to be verified. |
|
||||
|
|
|
|||
|
|
@ -0,0 +1,23 @@
|
|||
# AWS Athena
|
||||
|
||||

|
||||
|
||||
## Datasource Parameters
|
||||
|
||||
| **Datasource** | **Description** |
|
||||
|----------------------------|-----------------------------------------------------------|
|
||||
| Datasource | Select ATHENA. |
|
||||
| Datasource name | Enter the name of the DataSource. |
|
||||
| Description | Enter a description of the DataSource. |
|
||||
| Username | Set the AWS access key. |
|
||||
| Password | Set the AWS secret access key. |
|
||||
| AwsRegion | Set the AWS region. |
|
||||
| Database name | Enter the database name of the ATHENA connection. |
|
||||
| Jdbc connection parameters | Parameter settings for ATHENA connection, in JSON format. |
|
||||
|
||||
## Native Supported
|
||||
|
||||
- No, read section example in [datasource-setting](../howto/datasource-setting.md) `DataSource Center` section to activate this datasource.
|
||||
- JDBC driver configuration reference document [athena-connect-with-jdbc](https://docs.amazonaws.cn/athena/latest/ug/connect-with-jdbc.html)
|
||||
- Driver download link [SimbaAthenaJDBC-2.0.31.1000/AthenaJDBC42.jar](https://s3.cn-north-1.amazonaws.com.cn/athena-downloads-cn/drivers/JDBC/SimbaAthenaJDBC-2.0.31.1000/AthenaJDBC42.jar)
|
||||
|
||||
|
|
@ -5,7 +5,7 @@
|
|||
## Datasource Parameters
|
||||
|
||||
| **Datasource** | **Description** |
|
||||
| --- | --- |
|
||||
|-------------------------|---------------------------------------------------------------|
|
||||
| Datasource | Select CLICKHOUSE. |
|
||||
| Datasource Name | Enter the name of the datasource. |
|
||||
| Description | Enter a description of the datasource. |
|
||||
|
|
|
|||
|
|
@ -5,7 +5,7 @@
|
|||
## Datasource Parameters
|
||||
|
||||
| **Datasource** | **Description** |
|
||||
| --- | --- |
|
||||
|-------------------------|--------------------------------------------------------|
|
||||
| Datasource | Select DB2. |
|
||||
| Datasource Name | Enter the name of the datasource. |
|
||||
| Description | Enter a description of the datasource. |
|
||||
|
|
|
|||
|
|
@ -7,7 +7,7 @@
|
|||
## Datasource Parameters
|
||||
|
||||
| **Datasource** | **Description** |
|
||||
| --- | --- |
|
||||
|----------------------------|---------------------------------------------------------|
|
||||
| Datasource | Select HIVE. |
|
||||
| Datasource name | Enter the name of the DataSource. |
|
||||
| Description | Enter a description of the DataSource. |
|
||||
|
|
|
|||
|
|
@ -5,7 +5,7 @@
|
|||
## Datasource Parameters
|
||||
|
||||
| **Datasource** | **Description** |
|
||||
| --- | --- |
|
||||
|----------------------------|----------------------------------------------------------|
|
||||
| Datasource | Select MYSQL. |
|
||||
| Datasource name | Enter the name of the DataSource. |
|
||||
| Description | Enter a description of the DataSource. |
|
||||
|
|
|
|||
|
|
@ -5,7 +5,7 @@
|
|||
## Datasource Parameters
|
||||
|
||||
| **Datasource** | **Description** |
|
||||
| --- | --- |
|
||||
|-------------------------|---------------------------------------------------------------------------------------------------------------------------------------------|
|
||||
| Datasource | Select Oracle. |
|
||||
| Datasource Name | Enter the name of the datasource. |
|
||||
| Description | Enter a description of the datasource. |
|
||||
|
|
@ -13,8 +13,9 @@
|
|||
| Port | Enter the Oracle service port. |
|
||||
| Username | Set the username for Oracle connection. |
|
||||
| Password | Set the password for Oracle connection. |
|
||||
| Database Name | Enter the database name of the Oracle connection. |
|
||||
| jdbc connect parameters | Parameter settings for Oracle connection, in JSON format. |
|
||||
| Database Name | Enter the ServiceName or SID of the Oracle connection. |
|
||||
| ServiceName or SID | Choose ServiceName or SID according to your entry in Database Name column. |
|
||||
| jdbc connect parameters | Parameter settings for Oracle connection, in JSON format. For example, you can use {"schema": "abc"} to specify database abc for using. |
|
||||
|
||||
## Native Supported
|
||||
|
||||
|
|
|
|||
|
|
@ -5,7 +5,7 @@
|
|||
## Datasource Parameters
|
||||
|
||||
| **Datasource** | **Description** |
|
||||
| --- | --- |
|
||||
|----------------------------|---------------------------------------------------------------|
|
||||
| Datasource | Select POSTGRESQL. |
|
||||
| Datasource name | Enter the name of the DataSource. |
|
||||
| Description | Enter a description of the DataSource. |
|
||||
|
|
|
|||
|
|
@ -5,7 +5,7 @@
|
|||
## Datasource Parameters
|
||||
|
||||
| **Datasource** | **Description** |
|
||||
| --- | --- |
|
||||
|-------------------------|-----------------------------------------------------------|
|
||||
| Datasource | Select Presto. |
|
||||
| Datasource Name | Enter the name of the datasource. |
|
||||
| Description | Enter a description of the datasource. |
|
||||
|
|
@ -16,7 +16,6 @@
|
|||
| Database Name | Enter the database name of the Presto connection. |
|
||||
| jdbc connect parameters | Parameter settings for Presto connection, in JSON format. |
|
||||
|
||||
|
||||
## Native Supported
|
||||
|
||||
Yes, could use this datasource by default.
|
||||
|
|
@ -5,7 +5,7 @@
|
|||
## Datasource Parameters
|
||||
|
||||
| **Datasource** | **Description** |
|
||||
| --- | --- |
|
||||
|-------------------------|-------------------------------------------------------------|
|
||||
| Datasource | Select Redshift. |
|
||||
| Datasource Name | Enter the name of the datasource. |
|
||||
| Description | Enter a description of the datasource. |
|
||||
|
|
|
|||
|
|
@ -5,7 +5,7 @@
|
|||
## Datasource Parameters
|
||||
|
||||
| **Datasource** | **Description** |
|
||||
| --- | --- |
|
||||
|----------------------------|----------------------------------------------------------|
|
||||
| Datasource | Select Spark. |
|
||||
| Datasource name | Enter the name of the DataSource. |
|
||||
| Description | Enter a description of the DataSource. |
|
||||
|
|
|
|||
|
|
@ -5,7 +5,7 @@
|
|||
## Datasource Parameters
|
||||
|
||||
| **Datasource** | **Description** |
|
||||
| --- | --- |
|
||||
|-------------------------|--------------------------------------------------------------|
|
||||
| Datasource | Select SQLSERVER. |
|
||||
| Datasource Name | Enter the name of the datasource. |
|
||||
| Description | Enter a description of the datasource. |
|
||||
|
|
|
|||
|
|
@ -14,7 +14,6 @@ This article describes how to add a new master service or worker service to an e
|
|||
* [required] [JDK](https://www.oracle.com/technetwork/java/javase/downloads/index.html) (version 1.8+): must install, install and configure `JAVA_HOME` and `PATH` variables under `/etc/profile`
|
||||
* [optional] If the expansion is a worker node, you need to consider whether to install an external client, such as Hadoop, Hive, Spark Client.
|
||||
|
||||
|
||||
```markdown
|
||||
Attention: DolphinScheduler itself does not depend on Hadoop, Hive, Spark, but will only call their Client for the corresponding task submission.
|
||||
```
|
||||
|
|
@ -31,9 +30,9 @@ This article describes how to add a new master service or worker service to an e
|
|||
mkdir -p /opt
|
||||
cd /opt
|
||||
# decompress
|
||||
tar -zxvf apache-dolphinscheduler-<version>-bin.tar.gz -C /opt
|
||||
tar -zxvf apache-dolphinscheduler-3.1.1-bin.tar.gz -C /opt
|
||||
cd /opt
|
||||
mv apache-dolphinscheduler-<version>-bin dolphinscheduler
|
||||
mv apache-dolphinscheduler-3.1.1-bin dolphinscheduler
|
||||
```
|
||||
|
||||
```markdown
|
||||
|
|
@ -74,8 +73,7 @@ sed -i 's/Defaults requirett/#Defaults requirett/g' /etc/sudoers
|
|||
zookeeper.properties: information for connecting zk
|
||||
common.properties: Configuration information about the resource store (if hadoop is set up, please check if the core-site.xml and hdfs-site.xml configuration files exist).
|
||||
dolphinscheduler_env.sh: environment Variables
|
||||
````
|
||||
|
||||
```
|
||||
- Modify the `dolphinscheduler_env.sh` environment variable in the `bin/env/dolphinscheduler_env.sh` directory according to the machine configuration (the following is the example that all the used software install under `/opt/soft`)
|
||||
|
||||
```shell
|
||||
|
|
@ -94,15 +92,12 @@ sed -i 's/Defaults requirett/#Defaults requirett/g' /etc/sudoers
|
|||
|
||||
`Attention: This step is very important, such as `JAVA_HOME` and `PATH` is necessary to configure if haven not used just ignore or comment out`
|
||||
|
||||
|
||||
- Soft link the `JDK` to `/usr/bin/java` (still using `JAVA_HOME=/opt/soft/java` as an example)
|
||||
|
||||
```shell
|
||||
sudo ln -s /opt/soft/java/bin/java /usr/bin/java
|
||||
```
|
||||
|
||||
- Modify the configuration file `conf/config/install_config.conf` on the **all** nodes, synchronizing the following configuration.
|
||||
|
||||
* To add a new master node, you need to modify the IPs and masters parameters.
|
||||
* To add a new worker node, modify the IPs and workers parameters.
|
||||
|
||||
|
|
@ -120,6 +115,7 @@ masters="existing master01,existing master02,ds1,ds2"
|
|||
workers="existing worker01:default,existing worker02:default,ds3:default,ds4:default"
|
||||
|
||||
```
|
||||
|
||||
- If the expansion is for worker nodes, you need to set the worker group, refer to the security of the [Worker grouping](./security.md)
|
||||
|
||||
- On all new nodes, change the directory permissions so that the deployment user has access to the DolphinScheduler directory
|
||||
|
|
@ -222,13 +218,12 @@ bash bin/dolphinscheduler-daemon.sh start alert-server # start alert service
|
|||
ApiApplicationServer ----- api service
|
||||
AlertServer ----- alert service
|
||||
```
|
||||
If the corresponding master service or worker service does not exist, then the master or worker service is successfully shut down.
|
||||
|
||||
If the corresponding master service or worker service does not exist, then the master or worker service is successfully shut down.
|
||||
|
||||
### Modify the Configuration File
|
||||
|
||||
- modify the configuration file `conf/config/install_config.conf` on the **all** nodes, synchronizing the following configuration.
|
||||
|
||||
* to scale down the master node, modify the IPs and masters parameters.
|
||||
* to scale down worker nodes, modify the IPs and workers parameters.
|
||||
|
||||
|
|
@ -246,3 +241,4 @@ masters="existing master01,existing master02,ds1,ds2"
|
|||
workers="existing worker01:default,existing worker02:default,ds3:default,ds4:default"
|
||||
|
||||
```
|
||||
|
||||
|
|
|
|||
|
|
@ -39,3 +39,4 @@ curl --request GET 'http://localhost:50053/actuator/health'
|
|||
```
|
||||
|
||||
> Notice: If you modify the default service port and address, you need to modify the IP+Port to the modified value.
|
||||
|
||||
|
|
|
|||
|
|
@ -5,7 +5,7 @@
|
|||
We here use MySQL as an example to illustrate how to configure an external database:
|
||||
|
||||
> NOTE: If you use MySQL, you need to manually download [mysql-connector-java driver][mysql] (8.0.16) and move it to the libs directory of DolphinScheduler
|
||||
which is `api-server/libs` and `alert-server/libs` and `master-server/libs` and `worker-server/libs`.
|
||||
> which is `api-server/libs` and `alert-server/libs` and `master-server/libs` and `worker-server/libs`.
|
||||
|
||||
* First of all, follow the instructions in [datasource-setting](datasource-setting.md) `Pseudo-Cluster/Cluster Initialize the Database` section to create and initialize database
|
||||
* Set the following environment variables in your terminal or modify the `bin/env/dolphinscheduler_env.sh` with your database username and password for `{user}` and `{password}`:
|
||||
|
|
@ -26,7 +26,6 @@ DolphinScheduler stores metadata in `relational database`. Currently, we support
|
|||
|
||||
> If you use MySQL, you need to manually download [mysql-connector-java driver][mysql] (8.0.16) and move it to the libs directory of DolphinScheduler which is `api-server/libs` and `alert-server/libs` and `master-server/libs` and `worker-server/libs`.
|
||||
|
||||
|
||||
For mysql 5.6 / 5.7
|
||||
|
||||
```shell
|
||||
|
|
@ -57,6 +56,7 @@ mysql> FLUSH PRIVILEGES;
|
|||
```
|
||||
|
||||
For PostgreSQL:
|
||||
|
||||
```shell
|
||||
# Use psql-tools to login PostgreSQL
|
||||
psql
|
||||
|
|
@ -75,6 +75,7 @@ pg_ctl reload
|
|||
Then, modify `./bin/env/dolphinscheduler_env.sh`, change {user} and {password} to what you set in the previous step.
|
||||
|
||||
For MySQL:
|
||||
|
||||
```shell
|
||||
# for mysql
|
||||
export DATABASE=${DATABASE:-mysql}
|
||||
|
|
@ -85,6 +86,7 @@ export SPRING_DATASOURCE_PASSWORD={password}
|
|||
```
|
||||
|
||||
For PostgreSQL:
|
||||
|
||||
```shell
|
||||
# for postgresql
|
||||
export DATABASE=${DATABASE:-postgresql}
|
||||
|
|
@ -125,3 +127,4 @@ like Docker.
|
|||
> But if you want to use MySQL as the metabase of DolphinScheduler, it only supports [8.0.16 and above](https:/ /repo1.maven.org/maven2/mysql/mysql-connector-java/8.0.16/mysql-connector-java-8.0.16.jar) version.
|
||||
|
||||
[mysql]: https://downloads.MySQL.com/archives/c-j/
|
||||
|
||||
|
|
|
|||
|
|
@ -94,7 +94,7 @@ The configuration file is `values.yaml`, and the [Appendix-Configuration](#appen
|
|||
## Support Matrix
|
||||
|
||||
| Type | Support | Notes |
|
||||
| ------------------------------------------------------------ | ------------ | ------------------------------------- |
|
||||
|--------------------------------------------------------------|--------------|---------------------------------------|
|
||||
| Shell | Yes | |
|
||||
| Python2 | Yes | |
|
||||
| Python3 | Indirect Yes | Refer to FAQ |
|
||||
|
|
@ -522,7 +522,7 @@ common:
|
|||
## Appendix-Configuration
|
||||
|
||||
| Parameter | Description | Default |
|
||||
| --------------------------------------------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------ | ----------------------------------------------------- |
|
||||
|----------------------------------------------------------------------|-------------------------------------------------------------------------------------------------------------------------------|---------------------------------------|
|
||||
| `timezone` | World time and date for cities in all time zones | `Asia/Shanghai` |
|
||||
| | | |
|
||||
| `image.repository` | Docker image repository for the DolphinScheduler | `apache/dolphinscheduler` |
|
||||
|
|
@ -547,13 +547,14 @@ common:
|
|||
| `externalDatabase.params` | If exists external PostgreSQL, and set `postgresql.enabled` value to false. DolphinScheduler's database params will use it | `characterEncoding=utf8` |
|
||||
| | | |
|
||||
| `zookeeper.enabled` | If not exists external ZooKeeper, by default, the DolphinScheduler will use a internal ZooKeeper | `true` |
|
||||
| `zookeeper.service.port` | The port of zookeeper | `2181` |
|
||||
| `zookeeper.fourlwCommandsWhitelist` | A list of comma separated Four Letter Words commands to use | `srvr,ruok,wchs,cons` |
|
||||
| `zookeeper.persistence.enabled` | Set `zookeeper.persistence.enabled` to `true` to mount a new volume for internal ZooKeeper | `false` |
|
||||
| `zookeeper.persistence.size` | `PersistentVolumeClaim` size | `20Gi` |
|
||||
| `zookeeper.persistence.storageClass` | ZooKeeper data persistent volume storage class. If set to "-", storageClassName: "", which disables dynamic provisioning | `-` |
|
||||
| `zookeeper.zookeeperRoot` | Specify dolphinscheduler root directory in ZooKeeper | `/dolphinscheduler` |
|
||||
| `externalZookeeper.zookeeperQuorum` | If exists external ZooKeeper, and set `zookeeper.enabled` value to false. Specify Zookeeper quorum | `127.0.0.1:2181` |
|
||||
| `externalZookeeper.zookeeperRoot` | If exists external ZooKeeper, and set `zookeeper.enabled` value to false. Specify dolphinscheduler root directory in Zookeeper | `/dolphinscheduler` |
|
||||
| `externalRegistry.registryPluginDir` | If exists external registry and set `zookeeper.enable` to `false`, specify the external registry plugin directory | `lib/plugin/registry` |
|
||||
| `externalRegistry.registryPluginName` | If exists external registry and set `zookeeper.enable` to `false`, specify the external registry plugin name | `zookeeper` |
|
||||
| `externalRegistry.registryServers` | If exists external registry and set `zookeeper.enable` to `false`, specify the external registry servers | `127.0.0.1:2181` |
|
||||
| | | |
|
||||
| `common.configmap.DOLPHINSCHEDULER_OPTS` | The jvm options for dolphinscheduler, suitable for all servers | `""` |
|
||||
| `common.configmap.DATA_BASEDIR_PATH` | User data directory path, self configuration, please make sure the directory exists and have read write permissions | `/tmp/dolphinscheduler` |
|
||||
|
|
@ -744,3 +745,4 @@ common:
|
|||
| `ingress.path` | Ingress path | `/dolphinscheduler` |
|
||||
| `ingress.tls.enabled` | Enable ingress tls | `false` |
|
||||
| `ingress.tls.secretName` | Ingress tls secret name | `dolphinscheduler-tls` |
|
||||
|
||||
|
|
|
|||
|
|
@ -154,7 +154,7 @@ bash ./bin/install.sh
|
|||
```
|
||||
|
||||
> **_Note:_** For the first time deployment, there maybe occur five times of `sh: bin/dolphinscheduler-daemon.sh: No such file or directory` in the terminal,
|
||||
this is non-important information that you can ignore.
|
||||
> this is non-important information that you can ignore.
|
||||
|
||||
## Login DolphinScheduler
|
||||
|
||||
|
|
@ -190,7 +190,7 @@ bash ./bin/dolphinscheduler-daemon.sh stop alert-server
|
|||
> for micro-services need. It means that you could start all servers by command `<service>/bin/start.sh` with different
|
||||
> environment variable from `<service>/conf/dolphinscheduler_env.sh`. But it will use file `bin/env/dolphinscheduler_env.sh` overwrite
|
||||
> `<service>/conf/dolphinscheduler_env.sh` if you start server with command `/bin/dolphinscheduler-daemon.sh start <service>`.
|
||||
|
||||
>
|
||||
> **_Note2:_**: Please refer to the section of "System Architecture Design" for service usage. Python gateway service is
|
||||
> started along with the api-server, and if you do not want to start Python gateway service please disabled it by changing
|
||||
> the yaml config `python-gateway.enabled : false` in api-server's configuration path `api-server/conf/application.yaml`
|
||||
|
|
@ -198,3 +198,4 @@ bash ./bin/dolphinscheduler-daemon.sh stop alert-server
|
|||
[jdk]: https://www.oracle.com/technetwork/java/javase/downloads/index.html
|
||||
[zookeeper]: https://zookeeper.apache.org/releases.html
|
||||
[issue]: https://github.com/apache/dolphinscheduler/issues/6597
|
||||
|
||||
|
|
|
|||
|
|
@ -15,7 +15,7 @@ This section describes the one-click deployment of high availability DolphinSche
|
|||
* Click `install` on the right side of DolphinScheduler to go to the installation page. Fill in the corresponding information and click `OK` to start the installation. You will get automatically redirected to the application view.
|
||||
|
||||
| Select item | Description |
|
||||
| ------------ | ------------------------------------ |
|
||||
|--------------|-------------------------------------|
|
||||
| Team name | user workspace,Isolate by namespace |
|
||||
| Cluster name | select kubernetes cluster |
|
||||
| Select app | select application |
|
||||
|
|
@ -42,6 +42,7 @@ Take `worker` as an example: enter the `component -> Telescopic` page, and set t
|
|||
To verify `worker` node, enter `DolphinScheduler UI -> Monitoring -> Worker` page to view detailed node information.
|
||||
|
||||

|
||||
|
||||
## Configuration file
|
||||
|
||||
API and Worker Services share the configuration file `/opt/dolphinscheduler/conf/common.properties`. To modify the configurations, you only need to modify that of the API service.
|
||||
|
|
@ -61,4 +62,6 @@ Take `DataX` as an example:
|
|||
* LOCK_PATH:/opt/soft
|
||||
3. Update component, the plug-in `Datax` will be downloaded automatically and decompress to `/opt/soft`
|
||||

|
||||
|
||||
---
|
||||
|
||||
|
|
|
|||
|
|
@ -78,7 +78,6 @@ For example, you can get the master metrics by `curl http://localhost:5679/actua
|
|||
- ds.task.execution.count: (counter) the number of executed tasks
|
||||
- ds.task.execution.duration: (histogram) duration of task executions
|
||||
|
||||
|
||||
### Workflow Related Metrics
|
||||
|
||||
- ds.workflow.create.command.count: (counter) the number of commands created and inserted by workflows
|
||||
|
|
@ -175,3 +174,4 @@ For example, you can get the master metrics by `curl http://localhost:5679/actua
|
|||
- system.load.average.1m: the total number of runnable entities queued to available processors and runnable entities running on the available processors averaged over a period
|
||||
- logback.events: the number of events that made it to the logs grouped by the tag `level`
|
||||
- http.server.requests: total number of http requests
|
||||
|
||||
|
|
|
|||
|
|
@ -29,7 +29,7 @@
|
|||

|
||||
|
||||
| **Parameter** | **Description** |
|
||||
| ----- | ----- |
|
||||
|----------------------------------------|----------------------------------------------------|
|
||||
| Number of commands wait to be executed | Statistics of the `t_ds_command` table data. |
|
||||
| The number of failed commands | Statistics of the `t_ds_error_command` table data. |
|
||||
| Number of tasks wait to run | Count the data of `task_queue` in the ZooKeeper. |
|
||||
|
|
|
|||
|
|
@ -29,7 +29,7 @@ Generally, projects and processes are created through pages, but considering the
|
|||
2. select a test API, the API selected for this test is `queryAllProjectList`
|
||||
|
||||
> projects/list
|
||||
>
|
||||
|
||||
3. Open `Postman`, fill in the API address, enter the `Token` in `Headers`, and then send the request to view the result:
|
||||
|
||||
```
|
||||
|
|
|
|||
|
|
@ -3,7 +3,7 @@
|
|||
## Basic Built-in Parameter
|
||||
|
||||
| Variable | Declaration Method | Meaning |
|
||||
| ---- | ---- | -----------------------------|
|
||||
|--------------------|-------------------------|---------------------------------------------------------------------------------------------|
|
||||
| system.biz.date | `${system.biz.date}` | The day before the schedule time of the daily scheduling instance, the format is `yyyyMMdd` |
|
||||
| system.biz.curdate | `${system.biz.curdate}` | The schedule time of the daily scheduling instance, the format is `yyyyMMdd` |
|
||||
| system.datetime | `${system.datetime}` | The schedule time of the daily scheduling instance, the format is `yyyyMMddHHmmss` |
|
||||
|
|
@ -22,7 +22,6 @@
|
|||
- N years before:`$[add_months(yyyyMMdd,-12*N)]`
|
||||
- Next N months:`$[add_months(yyyyMMdd,N)]`
|
||||
- N months before:`$[add_months(yyyyMMdd,-N)]`
|
||||
|
||||
2. Add or minus numbers directly after the time format.
|
||||
- Next N weeks:`$[yyyyMMdd+7*N]`
|
||||
- First N weeks:`$[yyyyMMdd-7*N]`
|
||||
|
|
@ -32,3 +31,4 @@
|
|||
- First N hours:`$[HHmmss-N/24]`
|
||||
- Next N minutes:`$[HHmmss+N/24/60]`
|
||||
- First N minutes:`$[HHmmss-N/24/60]`
|
||||
|
||||
|
|
|
|||
|
|
@ -3,7 +3,7 @@
|
|||
This page describes details regarding Project screen in Apache DolphinScheduler. Here, you will see all the functions which can be handled in this screen. The following table explains commonly used terms in Apache DolphinScheduler:
|
||||
|
||||
| Glossary | description |
|
||||
| ------ |---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|
|
||||
|---------------------|---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|
|
||||
| DAG | Tasks in a workflow are assembled in form of Directed Acyclic Graph (DAG). A topological traversal is performed from nodes with zero degrees of entry until there are no subsequent nodes. |
|
||||
| Workflow Definition | Visualization formed by dragging task nodes and establishing task node associations (DAG). |
|
||||
| Workflow Instance | Instantiation of the workflow definition, which can be generated by manual start or scheduled scheduling. Each time the process definition runs, a workflow instance is generated. |
|
||||
|
|
|
|||
|
|
@ -1,6 +1,7 @@
|
|||
# Task Definition
|
||||
|
||||
## Batch Task Definition
|
||||
|
||||
Task definition allows to modify or operate tasks at the task level rather than modifying them in the workflow definition.
|
||||
We already have workflow level task editor in [workflow definition](workflow-definition.md) which you can click the specific
|
||||
workflow and then edit its task definition. It is depressing when you want to edit the task definition but do not remember
|
||||
|
|
@ -14,6 +15,7 @@ name but forget which workflow it belongs to. It is also supported query by the
|
|||
`Workflow Name`
|
||||
|
||||
## Stream Task Definition
|
||||
|
||||
Stream task definitions are created in the workflow definition, and can be modified and executed.
|
||||
|
||||

|
||||
|
|
|
|||
|
|
@ -1,6 +1,7 @@
|
|||
# Task Instance
|
||||
|
||||
## Batch Task Instance
|
||||
|
||||
### Create Task Instance
|
||||
|
||||
Click `Project Management -> Workflow -> Task Instance` to enter the task instance page, as shown in the figure below, click the name of the workflow instance to jump to the DAG diagram of the workflow instance to view the task status.
|
||||
|
|
@ -21,3 +22,4 @@ Click the `View Log` button in the operation column to view the log of the task
|
|||
|
||||
- SavePoint: Click the `SavePoint` button in the operation column to do stream task savepoint.
|
||||
- Stop: Click the `Stop` button in the operation column to stop the stream task.
|
||||
|
||||
|
|
|
|||
|
|
@ -35,6 +35,7 @@ Click the plus sign on the right of the task node to connect the task; as shown
|
|||

|
||||
|
||||
### Dependencies with stream task
|
||||
|
||||
If the DAG contains stream tasks, the relationship between stream tasks is displayed as a dotted line, and the execution of stream tasks will be skipped when the workflow instance is executed.
|
||||
|
||||

|
||||
|
|
@ -49,6 +50,17 @@ Click the `Save` button, and the "Set DAG chart name" window pops up, as shown i
|
|||
|
||||

|
||||
|
||||
### Configure workflow (process) execution type
|
||||
|
||||
Click the `Save` button and configure `process execution type` in the pop-up window. There are four process execution types:
|
||||
|
||||
- `Parallel`: If there are multiple instances of the same workflow definition, execute the instances in parallel.
|
||||
- `Serial Wait`: If there are multiple instances of the same workflow definition, execute the instances in serial.
|
||||
- `Serial Discard`: If there are multiple instances of the same workflow definition, discard the later ones and kill the current running ones.
|
||||
- `Serial Priority`: If there are multiple instances of the same workflow definition, execute the instances according to the priority in serial.
|
||||
|
||||

|
||||
|
||||
## Workflow Definition Operation Function
|
||||
|
||||
Click `Project Management -> Workflow -> Workflow Definition` to enter the workflow definition page, as shown below:
|
||||
|
|
@ -59,14 +71,50 @@ Workflow running parameter description:
|
|||
|
||||
* **Failure strategy**: When a task node fails to execute, other parallel task nodes need to execute the strategy. "Continue" means: After a task fails, other task nodes execute normally; "End" means: Terminate all tasks being executed, and terminate the entire process.
|
||||
* **Notification strategy**: When the process ends, send process execution information notification emails according to the process status, including no status, success, failure, success or failure.
|
||||
* **Process priority**: the priority of process operation, divided into five levels: the highest (HIGHEST), high (HIGH), medium (MEDIUM), low (LOW), the lowest (LOWEST). When the number of master threads is insufficient, processes with higher levels will be executed first in the execution queue, and processes with the same priority will be executed in the order of first-in, first-out.
|
||||
* **Process priority**: The priority of process execution, there are five different priorities: the highest (HIGHEST), high (HIGH), medium (MEDIUM), low (LOW), the lowest (LOWEST). When the number of master threads is insufficient, processes with higher priorities in the execution queue will run first. Processes with the same priority will run in first-come-first-served fashion.
|
||||
* **Worker grouping**: This process can only be executed in the specified worker machine group. The default is Default, which can be executed on any worker.
|
||||
* **Notification Group**: Select Notification Policy||Timeout Alarm||When fault tolerance occurs, process information or emails will be sent to all members in the notification group.
|
||||
* **Recipient**: Select Notification Policy||Timeout Alarm||When fault tolerance occurs, process information or alarm email will be sent to the recipient list.
|
||||
* **Cc**: Select Notification Policy||Timeout Alarm||When fault tolerance occurs, the process information or alarm email will be copied to the Cc list.
|
||||
* **Startup parameters**: Set or override the value of global parameters when starting a new process instance.
|
||||
* **Complement**: There are 2 modes of serial complement and parallel complement. Serial complement: within the specified time range, perform complements in sequence from the start date to the end date, and generate N process instances in turn; parallel complement: within the specified time range, perform multiple complements at the same time, and generate N process instances at the same time .
|
||||
* **Complement**: Execute the workflow definition of the specified date, you can select the time range of the supplement (currently only supports the supplement for consecutive days), for example, the data from May 1st to May 10th needs to be supplemented, as shown in the following figure:
|
||||
* **Complement(Backfill)**: Run workflow for a specified historical period. There are two strategies: serial complement and parallel complement.
|
||||
> You could select the time period or fill in it manually in UI. The date range is left closed and right closed time interval (startDate <= N <= endDate)
|
||||
* Serial complement: Run the workflow from start date to end date according to the time period you set in serial.
|
||||
|
||||

|
||||
|
||||
* Parallel complement: Run the workflow from start date to end date according to the time period you set in parallel.
|
||||
|
||||

|
||||
|
||||
* Parallelism: The max number of workflow instances of the workflow definition you choose for complement.
|
||||

|
||||
|
||||

|
||||
|
||||
* Mode of dependent: Whether to trigger downstream workflow definition for complement.
|
||||
|
||||

|
||||
|
||||
* Schedule date:
|
||||
|
||||
1. Select from pop-up window:
|
||||
|
||||

|
||||
|
||||
2. Fill in the time period manually:
|
||||
|
||||

|
||||
|
||||
* Complement with or without scheduling:
|
||||
|
||||
1. `Unconfigured timing` or `Configured timing and timing status offline`: Complement the number according to the selected time range combined with the timing default configuration (0:00 every day). e.g. the workflow scheduling date is from July 7th to July 10th:
|
||||
|
||||

|
||||
|
||||
2. `Configured timing and timing status online`: Complement the number according to the selected time range combined with the timing configuration. e.g. the workflow scheduling date is from July 7th to July 10th, and the timing is configured (running at 5 am every day):
|
||||
|
||||

|
||||
|
||||

|
||||
|
||||
The following are the operation functions of the workflow definition list:
|
||||
|
||||
|
|
@ -103,7 +151,6 @@ The following are the operation functions of the workflow definition list:
|
|||
* Cc: select notification policy||timeout alarm||when fault tolerance occurs, the process result information or warning email will be copied to the CC list.
|
||||
* Startup parameter: Set or overwrite global parameter values when starting a new process instance.
|
||||
* Complement: refers to running the workflow definition within the specified date range and generating the corresponding workflow instance according to the complement policy. The complement policy includes two modes: **serial complement** and **parallel complement**. The date can be selected on the page or entered manually.
|
||||
|
||||
* Serial complement: within the specified time range, complement is executed from the start date to the end date, and multiple process instances are generated in turn; Click Run workflow and select the serial complement mode: for example, from July 9 to July 10, execute in sequence, and generate two process instances in sequence on the process instance page.
|
||||
|
||||

|
||||
|
|
@ -143,6 +190,7 @@ The following are the operation functions of the workflow definition list:
|
|||

|
||||
|
||||

|
||||
|
||||
## Run the task alone
|
||||
|
||||
- Right-click the task and click the `Start` button (only online tasks can be clicked to run).
|
||||
|
|
@ -160,12 +208,15 @@ The following are the operation functions of the workflow definition list:
|
|||

|
||||
|
||||
- Select a start and end time. Within the start and end time range, the workflow is run regularly; outside the start and end time range, no timed workflow instance will be generated.
|
||||
|
||||
- Add a timing that execute 5 minutes once, as shown in the following figure:
|
||||
|
||||

|
||||
|
||||
- Failure strategy, notification strategy, process priority, worker group, notification group, recipient, and CC are the same as workflow running parameters.
|
||||
|
||||
- Click the "Create" button to create the timing. Now the timing status is "**Offline**" and the timing needs to be **Online** to make effect.
|
||||
|
||||
- Schedule online: Click the `Timing Management` button <img src="../../../../img/timeManagement.png" width="35"/>, enter the timing management page, click the `online` button, the timing status will change to `online`, as shown in the below figure, the workflow makes effect regularly.
|
||||
|
||||

|
||||
|
|
|
|||
|
|
@ -43,15 +43,23 @@ Click `Project Management -> Workflow -> Workflow Instance`, enter the workflow
|
|||

|
||||
|
||||
- **Edit:** Only processes with success/failed/stop status can be edited. Click the "Edit" button or the workflow instance name to enter the DAG edit page. After the edit, click the "Save" button to confirm, as shown in the figure below. In the pop-up box, check "Whether to update the workflow definition", after saving, the information modified by the instance will be updated to the workflow definition; if not checked, the workflow definition would not be updated.
|
||||
|
||||
<p align="center">
|
||||
<img src="../../../../img/editDag-en.png" width="80%" />
|
||||
</p>
|
||||
|
||||
- **Rerun:** Re-execute the terminated process
|
||||
|
||||
- **Recovery Failed:** For failed processes, you can perform failure recovery operations, starting from the failed node
|
||||
|
||||
- **Stop:** **Stop** the running process, the background code will first `kill` the worker process, and then execute `kill -9` operation
|
||||
|
||||
- **Pause:** **Pause** the running process, the system status will change to **waiting for execution**, it will wait for the task to finish, and pause the next sequence task.
|
||||
|
||||
- **Resume pause:** Resume the paused process, start running directly from the **paused node**
|
||||
|
||||
- **Delete:** Delete the workflow instance and the task instance under the workflow instance
|
||||
|
||||
- **Gantt Chart:** The vertical axis of the Gantt chart is the topological sorting of task instances of the workflow instance, and the horizontal axis is the running time of the task instances, as shown in the figure:
|
||||
|
||||

|
||||
|
|
|
|||
|
|
@ -45,9 +45,9 @@ data.basedir.path=/tmp/dolphinscheduler
|
|||
# resource view suffixs
|
||||
#resource.view.suffixs=txt,log,sh,bat,conf,cfg,py,java,sql,xml,hql,properties,json,yml,yaml,ini,js
|
||||
|
||||
# resource storage type: HDFS, S3, NONE
|
||||
# resource storage type: HDFS, S3, OSS, NONE
|
||||
resource.storage.type=NONE
|
||||
# resource store on HDFS/S3 path, resource file will store to this base path, self configuration, please make sure the directory exists on hdfs and have read write permissions. "/dolphinscheduler" is recommended
|
||||
# resource store on HDFS/S3/OSS path, resource file will store to this base path, self configuration, please make sure the directory exists on hdfs and have read write permissions. "/dolphinscheduler" is recommended
|
||||
resource.storage.upload.base.path=/tmp/dolphinscheduler
|
||||
|
||||
# The AWS access key. if resource.storage.type=S3 or use EMR-Task, This configuration is required
|
||||
|
|
@ -61,6 +61,17 @@ resource.aws.s3.bucket.name=dolphinscheduler
|
|||
# You need to set this parameter when private cloud s3. If S3 uses public cloud, you only need to set resource.aws.region or set to the endpoint of a public cloud such as S3.cn-north-1.amazonaws.com.cn
|
||||
resource.aws.s3.endpoint=http://localhost:9000
|
||||
|
||||
# alibaba cloud access key id, required if you set resource.storage.type=OSS
|
||||
resource.alibaba.cloud.access.key.id=<your-access-key-id>
|
||||
# alibaba cloud access key secret, required if you set resource.storage.type=OSS
|
||||
resource.alibaba.cloud.access.key.secret=<your-access-key-secret>
|
||||
# alibaba cloud region, required if you set resource.storage.type=OSS
|
||||
resource.alibaba.cloud.region=cn-hangzhou
|
||||
# oss bucket name, required if you set resource.storage.type=OSS
|
||||
resource.alibaba.cloud.oss.bucket.name=dolphinscheduler
|
||||
# oss bucket endpoint, required if you set resource.storage.type=OSS
|
||||
resource.alibaba.cloud.oss.endpoint=https://oss-cn-hangzhou.aliyuncs.com
|
||||
|
||||
# if resource.storage.type=HDFS, the user must have the permission to create directories under the HDFS root path
|
||||
resource.hdfs.root.user=hdfs
|
||||
# if resource.storage.type=S3, the value like: s3a://dolphinscheduler; if resource.storage.type=HDFS and namenode HA is enabled, you need to copy core-site.xml and hdfs-site.xml to conf dir
|
||||
|
|
|
|||
|
|
@ -65,6 +65,7 @@ In the workflow definition module of project Manage, create a new workflow using
|
|||
|
||||
- Script: 'sh hello.sh'
|
||||
- Resource: Select 'hello.sh'
|
||||
|
||||
> Notice: When using a resource file in the script, the file name needs to be the same as the full path of the selected resource:
|
||||
> For example: if the resource path is `/resource/hello.sh`, you need to use the full path of `/resource/hello.sh` to use it in the script.
|
||||
|
||||
|
|
|
|||
Some files were not shown because too many files have changed in this diff Show More
Loading…
Reference in New Issue