diff --git a/pom.xml b/pom.xml index 9f90652f93..c20ee08d7f 100644 --- a/pom.xml +++ b/pom.xml @@ -125,6 +125,7 @@ 3.12.0 1.6.1 3.23.1 + 1.29.0 3.10.1 3.2.4 @@ -142,7 +143,7 @@ 3.3.0 1.6.0 org.apache.streampark.shaded - 5.1 + 5.4.2 1.18.24 5.9.1 3.4.6 @@ -342,6 +343,12 @@ ${commons-collections4.version} + + io.openlineage + openlineage-java + ${openlineage.version} + + org.apache.httpcomponents.client5 httpclient5 @@ -354,6 +361,21 @@ ${httpclient5.version} + + + org.apache.httpcomponents.core5 + httpcore5 + 5.4.3 + + + + org.apache.httpcomponents.core5 + httpcore5-h2 + 5.4.3 + + org.apache.hadoop hadoop-client-api diff --git a/streampark-console/streampark-console-service/pom.xml b/streampark-console/streampark-console-service/pom.xml index b82cd03e55..a865276c29 100644 --- a/streampark-console/streampark-console-service/pom.xml +++ b/streampark-console/streampark-console-service/pom.xml @@ -295,6 +295,12 @@ ${commons-email.version} + + + io.openlineage + openlineage-java + + com.auth0 diff --git a/streampark-console/streampark-console-service/src/main/assembly/script/data/mysql-data.sql b/streampark-console/streampark-console-service/src/main/assembly/script/data/mysql-data.sql index 2e87336901..ba7524b5fe 100644 --- a/streampark-console/streampark-console-service/src/main/assembly/script/data/mysql-data.sql +++ b/streampark-console/streampark-console-service/src/main/assembly/script/data/mysql-data.sql @@ -328,6 +328,10 @@ insert into `t_setting` values (12, 'docker.register.user', null, 'Docker Regist insert into `t_setting` values (13, 'docker.register.password', null, 'Docker Register Password', 'Docker container service authentication password', 1); insert into `t_setting` values (14, 'docker.register.namespace', null, 'Docker namespace', 'Namespace for docker image used in docker building env and target image register', 1); insert into `t_setting` values (15, 'ingress.mode.default', null, 'Ingress domain address', 'Automatically generate an nginx-based ingress by passing in a domain name', 1); +insert into `t_setting` values (16, 'lineage.gravitino.address', null, 'Gravitino Address', 'Base URL of the Gravitino server lineage events are reported to, e.g. http://host:8090', 1); +insert into `t_setting` values (17, 'lineage.gravitino.token', null, 'Gravitino Auth Token', 'Bearer token forwarded to Gravitino, required once Gravitino oauth authentication is enabled', 1); +insert into `t_setting` values (18, 'lineage.gravitino.namespace', null, 'Gravitino Lineage Namespace', 'OpenLineage job/dataset namespace StreamPark reports under', 1); +insert into `t_setting` values (19, 'lineage.flink.native.listener.enable', 'true', 'Enable Flink Native OpenLineage Listener', 'Whether to also inject the official openlineage-flink job-status-changed-listener config; only takes effect once Gravitino Address is set', 2); -- ---------------------------- -- Records of t_user -- ---------------------------- diff --git a/streampark-console/streampark-console-service/src/main/assembly/script/data/pgsql-data.sql b/streampark-console/streampark-console-service/src/main/assembly/script/data/pgsql-data.sql index a3f2d6c354..9c9da60442 100644 --- a/streampark-console/streampark-console-service/src/main/assembly/script/data/pgsql-data.sql +++ b/streampark-console/streampark-console-service/src/main/assembly/script/data/pgsql-data.sql @@ -277,6 +277,10 @@ insert into "public"."t_setting" values (12, 'docker.register.user', null, 'Dock insert into "public"."t_setting" values (13, 'docker.register.password', null, 'Docker Register Password', 'Docker container service authentication password', 1); insert into "public"."t_setting" values (14, 'docker.register.namespace', null, 'Docker namespace', 'Namespace for docker image used in docker building env and target image register', 1); insert into "public"."t_setting" values (15, 'ingress.mode.default', null, 'Ingress domain address', 'Automatically generate an nginx-based ingress by passing in a domain name', 1); +insert into "public"."t_setting" values (16, 'lineage.gravitino.address', null, 'Gravitino Address', 'Base URL of the Gravitino server lineage events are reported to, e.g. http://host:8090', 1); +insert into "public"."t_setting" values (17, 'lineage.gravitino.token', null, 'Gravitino Auth Token', 'Bearer token forwarded to Gravitino, required once Gravitino oauth authentication is enabled', 1); +insert into "public"."t_setting" values (18, 'lineage.gravitino.namespace', null, 'Gravitino Lineage Namespace', 'OpenLineage job/dataset namespace StreamPark reports under', 1); +insert into "public"."t_setting" values (19, 'lineage.flink.native.listener.enable', 'true', 'Enable Flink Native OpenLineage Listener', 'Whether to also inject the official openlineage-flink job-status-changed-listener config; only takes effect once Gravitino Address is set', 2); -- ---------------------------- -- Records of t_user diff --git a/streampark-console/streampark-console-service/src/main/assembly/script/schema/mysql-schema.sql b/streampark-console/streampark-console-service/src/main/assembly/script/schema/mysql-schema.sql index db29202ced..b8c72edc2b 100644 --- a/streampark-console/streampark-console-service/src/main/assembly/script/schema/mysql-schema.sql +++ b/streampark-console/streampark-console-service/src/main/assembly/script/schema/mysql-schema.sql @@ -99,6 +99,7 @@ create table `t_flink_app` ( `default_mode_ingress` text collate utf8mb4_general_ci, `tags` varchar(500) default null, `hadoop_user` varchar(64) collate utf8mb4_general_ci default null, + `lineage_enable` tinyint default 0, primary key (`id`) using btree, key `inx_job_type` (`job_type`) using btree, key `inx_track` (`tracking`) using btree, @@ -616,6 +617,7 @@ create table `t_spark_app` ( `k8s_executor_pod_template` text collate utf8mb4_general_ci, `k8s_hadoop_integration` tinyint default 0, `hadoop_user` varchar(64) collate utf8mb4_general_ci default null, + `lineage_enable` tinyint default 0, `restart_size` int default null, `restart_count` int default null, `state` int default null, diff --git a/streampark-console/streampark-console-service/src/main/assembly/script/schema/pgsql-schema.sql b/streampark-console/streampark-console-service/src/main/assembly/script/schema/pgsql-schema.sql index a638e7eebd..61ac36ac4f 100644 --- a/streampark-console/streampark-console-service/src/main/assembly/script/schema/pgsql-schema.sql +++ b/streampark-console/streampark-console-service/src/main/assembly/script/schema/pgsql-schema.sql @@ -252,7 +252,8 @@ create table "public"."t_flink_app" ( "ingress_template" text collate "pg_catalog"."default", "default_mode_ingress" text collate "pg_catalog"."default", "tags" varchar(500) collate "pg_catalog"."default", - "hadoop_user" varchar(63) collate "pg_catalog"."default" + "hadoop_user" varchar(63) collate "pg_catalog"."default", + "lineage_enable" boolean default false ) ; alter table "public"."t_flink_app" add constraint "t_flink_app_pkey" primary key ("id"); diff --git a/streampark-console/streampark-console-service/src/main/assembly/script/upgrade/mysql/3.0.0.sql b/streampark-console/streampark-console-service/src/main/assembly/script/upgrade/mysql/3.0.0.sql new file mode 100644 index 0000000000..3c9ff8ff1d --- /dev/null +++ b/streampark-console/streampark-console-service/src/main/assembly/script/upgrade/mysql/3.0.0.sql @@ -0,0 +1,38 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one or more + * contributor license agreements. See the NOTICE file distributed with + * this work for additional information regarding copyright ownership. + * The ASF licenses this file to You under the Apache License, Version 2.0 + * (the "License"); you may not use this file except in compliance with + * the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +use streampark; + +set names utf8mb4; +set foreign_key_checks = 0; + +-- ---------------------------- +-- lineage: t_setting +-- ---------------------------- +insert into `t_setting` values (16, 'lineage.gravitino.address', null, 'Gravitino Address', 'Base URL of the Gravitino server lineage events are reported to, e.g. http://host:8090', 1); +insert into `t_setting` values (17, 'lineage.gravitino.token', null, 'Gravitino Auth Token', 'Bearer token forwarded to Gravitino, required once Gravitino oauth authentication is enabled', 1); +insert into `t_setting` values (18, 'lineage.gravitino.namespace', null, 'Gravitino Lineage Namespace', 'OpenLineage job/dataset namespace StreamPark reports under', 1); +insert into `t_setting` values (19, 'lineage.flink.native.listener.enable', 'true', 'Enable Flink Native OpenLineage Listener', 'Whether to also inject the official openlineage-flink job-status-changed-listener config; only takes effect once Gravitino Address is set', 2); + +-- ---------------------------- +-- lineage: t_flink_app / t_spark_app +-- ---------------------------- +alter table `t_flink_app` +add column `lineage_enable` tinyint default 0; + +alter table `t_spark_app` +add column `lineage_enable` tinyint default 0; diff --git a/streampark-console/streampark-console-service/src/main/assembly/script/upgrade/pgsql/3.0.0.sql b/streampark-console/streampark-console-service/src/main/assembly/script/upgrade/pgsql/3.0.0.sql new file mode 100644 index 0000000000..4d5d9d6425 --- /dev/null +++ b/streampark-console/streampark-console-service/src/main/assembly/script/upgrade/pgsql/3.0.0.sql @@ -0,0 +1,35 @@ +/* +* Licensed to the Apache Software Foundation (ASF) under one or more +* contributor license agreements. See the NOTICE file distributed with +* this work for additional information regarding copyright ownership. +* The ASF licenses this file to You under the Apache License, Version 2.0 +* (the "License"); you may not use this file except in compliance with +* the License. You may obtain a copy of the License at +* +* http://www.apache.org/licenses/LICENSE-2.0 +* +* Unless required by applicable law or agreed to in writing, software +* distributed under the License is distributed on an "AS IS" BASIS, +* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +* See the License for the specific language governing permissions and +* limitations under the License. +*/ + +-- ---------------------------- +-- lineage: t_setting +-- ---------------------------- +insert into "public"."t_setting" values (16, 'lineage.gravitino.address', null, 'Gravitino Address', 'Base URL of the Gravitino server lineage events are reported to, e.g. http://host:8090', 1); +insert into "public"."t_setting" values (17, 'lineage.gravitino.token', null, 'Gravitino Auth Token', 'Bearer token forwarded to Gravitino, required once Gravitino oauth authentication is enabled', 1); +insert into "public"."t_setting" values (18, 'lineage.gravitino.namespace', null, 'Gravitino Lineage Namespace', 'OpenLineage job/dataset namespace StreamPark reports under', 1); +insert into "public"."t_setting" values (19, 'lineage.flink.native.listener.enable', 'true', 'Enable Flink Native OpenLineage Listener', 'Whether to also inject the official openlineage-flink job-status-changed-listener config; only takes effect once Gravitino Address is set', 2); + +-- ---------------------------- +-- lineage: t_flink_app +-- ---------------------------- +alter table "public"."t_flink_app" +add column "lineage_enable" boolean default false; + +-- Note: t_spark_app has no PostgreSQL schema definition anywhere in this repository (pre-existing +-- gap, confirmed absent from pgsql-schema.sql; not introduced by this change and not fixed here, +-- out of scope for a lineage feature migration). The corresponding "lineage_enable" column for +-- t_spark_app is added only in upgrade/mysql/3.0.0.sql; add it here once that gap is fixed. diff --git a/streampark-console/streampark-console-service/src/main/java/org/apache/streampark/console/core/bean/LineageConfig.java b/streampark-console/streampark-console-service/src/main/java/org/apache/streampark/console/core/bean/LineageConfig.java new file mode 100644 index 0000000000..780abe27bb --- /dev/null +++ b/streampark-console/streampark-console-service/src/main/java/org/apache/streampark/console/core/bean/LineageConfig.java @@ -0,0 +1,68 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one or more + * contributor license agreements. See the NOTICE file distributed with + * this work for additional information regarding copyright ownership. + * The ASF licenses this file to You under the Apache License, Version 2.0 + * (the "License"); you may not use this file except in compliance with + * the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +package org.apache.streampark.console.core.bean; + +import org.apache.commons.lang3.StringUtils; + +import lombok.Getter; +import lombok.Setter; + +/** + * Gravitino lineage configuration. {@link #enabled()} is the single gate every lineage call site + * checks before doing any work: it is only true when an operator has actually filled in a + * Gravitino address, so a fresh install with these settings left blank injects nothing and emits + * nothing. + */ +@Getter +@Setter +public class LineageConfig { + + /** + * Namespace used when an operator has not set one. It ends up inside emitted event payloads and + * inside injected job configuration, so both paths must agree on it — hence one definition. + */ + public static final String DEFAULT_NAMESPACE = "streampark"; + + /** + * Gravitino's lineage-ingest path, appended to {@link #gravitinoAddress}. Written into the + * Flink and Spark listener configuration as well as used by the Console's own emitter, and all + * three must address the same endpoint — hence one definition. + */ + public static final String LINEAGE_ENDPOINT_PATH = "/api/lineage"; + + /** Gravitino base URL, e.g. {@code http://192.168.10.132:8090}. */ + private String gravitinoAddress; + + /** Bearer token forwarded as-is to Gravitino's {@code /api/lineage}; required once oauth is enabled there. */ + private String gravitinoToken; + + /** OpenLineage job/dataset namespace StreamPark reports under. */ + private String gravitinoNamespace; + + /** Whether to also inject the official {@code openlineage-flink} job-status-changed-listener config. */ + private boolean flinkNativeListenerEnable; + + public boolean enabled() { + return StringUtils.isNotBlank(gravitinoAddress); + } + + /** The configured namespace, or {@link #DEFAULT_NAMESPACE} when none was set. */ + public String namespaceOrDefault() { + return StringUtils.defaultIfBlank(gravitinoNamespace, DEFAULT_NAMESPACE); + } +} diff --git a/streampark-console/streampark-console-service/src/main/java/org/apache/streampark/console/core/entity/FlinkApplication.java b/streampark-console/streampark-console-service/src/main/java/org/apache/streampark/console/core/entity/FlinkApplication.java index f16dfc675f..1595b48e74 100644 --- a/streampark-console/streampark-console-service/src/main/java/org/apache/streampark/console/core/entity/FlinkApplication.java +++ b/streampark-console/streampark-console-service/src/main/java/org/apache/streampark/console/core/entity/FlinkApplication.java @@ -140,6 +140,9 @@ public class FlinkApplication extends BaseEntity implements ApplicationEntitySup */ private Boolean k8sHadoopIntegration; + /** Whether to report OpenLineage data lineage for this application to Gravitino. */ + private Boolean lineageEnable; + private Integer state; /** * task release status diff --git a/streampark-console/streampark-console-service/src/main/java/org/apache/streampark/console/core/entity/SparkApplication.java b/streampark-console/streampark-console-service/src/main/java/org/apache/streampark/console/core/entity/SparkApplication.java index 2f59d1cd17..83402c845a 100644 --- a/streampark-console/streampark-console-service/src/main/java/org/apache/streampark/console/core/entity/SparkApplication.java +++ b/streampark-console/streampark-console-service/src/main/java/org/apache/streampark/console/core/entity/SparkApplication.java @@ -147,6 +147,9 @@ public class SparkApplication extends BaseEntity implements ApplicationEntitySup @TableField("HADOOP_USER") private String hadoopUser; + /** Whether to report OpenLineage data lineage for this application to Gravitino. */ + private Boolean lineageEnable; + /** max restart retries after job failed */ @TableField(updateStrategy = FieldStrategy.IGNORED) private Integer restartSize; diff --git a/streampark-console/streampark-console-service/src/main/java/org/apache/streampark/console/core/request/flink/FlinkAppCreateRequest.java b/streampark-console/streampark-console-service/src/main/java/org/apache/streampark/console/core/request/flink/FlinkAppCreateRequest.java index ce5839097a..0d967d4290 100644 --- a/streampark-console/streampark-console-service/src/main/java/org/apache/streampark/console/core/request/flink/FlinkAppCreateRequest.java +++ b/streampark-console/streampark-console-service/src/main/java/org/apache/streampark/console/core/request/flink/FlinkAppCreateRequest.java @@ -125,4 +125,6 @@ public class FlinkAppCreateRequest implements Serializable { private Boolean k8sHadoopIntegration; private String serviceAccount; + + private Boolean lineageEnable; } diff --git a/streampark-console/streampark-console-service/src/main/java/org/apache/streampark/console/core/request/spark/SparkAppCreateRequest.java b/streampark-console/streampark-console-service/src/main/java/org/apache/streampark/console/core/request/spark/SparkAppCreateRequest.java index ee73011990..b5a383c3a0 100644 --- a/streampark-console/streampark-console-service/src/main/java/org/apache/streampark/console/core/request/spark/SparkAppCreateRequest.java +++ b/streampark-console/streampark-console-service/src/main/java/org/apache/streampark/console/core/request/spark/SparkAppCreateRequest.java @@ -86,6 +86,8 @@ public class SparkAppCreateRequest implements Serializable { private String hadoopUser; + private Boolean lineageEnable; + private Integer restartSize; private Long alertId; diff --git a/streampark-console/streampark-console-service/src/main/java/org/apache/streampark/console/core/response/flink/FlinkAppResponse.java b/streampark-console/streampark-console-service/src/main/java/org/apache/streampark/console/core/response/flink/FlinkAppResponse.java index f3411f2a42..75cc17cb1e 100644 --- a/streampark-console/streampark-console-service/src/main/java/org/apache/streampark/console/core/response/flink/FlinkAppResponse.java +++ b/streampark-console/streampark-console-service/src/main/java/org/apache/streampark/console/core/response/flink/FlinkAppResponse.java @@ -152,6 +152,8 @@ public class FlinkAppResponse implements Serializable { private Boolean k8sHadoopIntegration; + private Boolean lineageEnable; + private JobsOverview.Task overview; private String teamResource; diff --git a/streampark-console/streampark-console-service/src/main/java/org/apache/streampark/console/core/response/spark/SparkAppResponse.java b/streampark-console/streampark-console-service/src/main/java/org/apache/streampark/console/core/response/spark/SparkAppResponse.java index 3d3c2f4569..f65da09575 100644 --- a/streampark-console/streampark-console-service/src/main/java/org/apache/streampark/console/core/response/spark/SparkAppResponse.java +++ b/streampark-console/streampark-console-service/src/main/java/org/apache/streampark/console/core/response/spark/SparkAppResponse.java @@ -91,6 +91,8 @@ public class SparkAppResponse implements Serializable { private String hadoopUser; + private Boolean lineageEnable; + private Integer restartSize; private Integer restartCount; diff --git a/streampark-console/streampark-console-service/src/main/java/org/apache/streampark/console/core/service/GravitinoLineageService.java b/streampark-console/streampark-console-service/src/main/java/org/apache/streampark/console/core/service/GravitinoLineageService.java new file mode 100644 index 0000000000..dc51b9b8fa --- /dev/null +++ b/streampark-console/streampark-console-service/src/main/java/org/apache/streampark/console/core/service/GravitinoLineageService.java @@ -0,0 +1,63 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one or more + * contributor license agreements. See the NOTICE file distributed with + * this work for additional information regarding copyright ownership. + * The ASF licenses this file to You under the Apache License, Version 2.0 + * (the "License"); you may not use this file except in compliance with + * the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +package org.apache.streampark.console.core.service; + +import org.apache.streampark.console.core.entity.FlinkApplication; +import org.apache.streampark.flink.core.lineage.LineagePipeline; + +import java.util.List; + +/** + * Reports Flink job table-level lineage to Gravitino's {@code POST /api/lineage} as OpenLineage + * {@code RunEvent}s. + * + *

Every method here is fail-open by contract: a disabled switch, an unconfigured Gravitino + * address, or any failure while talking to Gravitino is logged and swallowed, never thrown. This + * runs on the job submission and state-watching paths, where a lineage gap must never affect the + * job itself. + * + *

{@link #trackAndEmitStart} and {@link #emitTerminal} are a pair: a successful start call + * remembers the run in memory so the later terminal call (driven by {@code FlinkAppHttpWatcher}'s + * state polling) knows what to close out, without needing the caller to thread pipeline data + * through the whole state-watching path. This tracking is in-memory only — it does not survive a + * Console restart, so a run whose job finishes while Console is down never gets its COMPLETE/FAIL + * event. That is a deliberate, bounded scope decision (see the implementation), not an oversight. + */ +public interface GravitinoLineageService { + + /** + * Called once, right after a Flink SQL job's submission succeeds. No-ops if {@code pipelines} + * is empty (extraction found nothing, or lineage is disabled for this application/globally). + * + * @param application the just-started application (its id keys the in-memory pending-run + * tracking consumed by {@link #emitTerminal}) + * @param flinkJobIdHex the Flink JobID this run was submitted with + * @param pipelines the pipelines resolved from the job's SQL; safe to pass an empty list + */ + void trackAndEmitStart(FlinkApplication application, String flinkJobIdHex, List pipelines); + + /** + * Called when {@code FlinkAppHttpWatcher} observes an application transition into a terminal + * state. No-ops if no pending run is tracked for {@code appId} (lineage was never started for + * this run, or it was already closed out). + * + * @param appId the application id + * @param success {@code true} to emit COMPLETE, {@code false} to emit FAIL + */ + void emitTerminal(Long appId, boolean success); +} diff --git a/streampark-console/streampark-console-service/src/main/java/org/apache/streampark/console/core/service/SettingService.java b/streampark-console/streampark-console-service/src/main/java/org/apache/streampark/console/core/service/SettingService.java index e2f309d8f0..df9a0536d8 100644 --- a/streampark-console/streampark-console-service/src/main/java/org/apache/streampark/console/core/service/SettingService.java +++ b/streampark-console/streampark-console-service/src/main/java/org/apache/streampark/console/core/service/SettingService.java @@ -18,6 +18,7 @@ package org.apache.streampark.console.core.service; import org.apache.streampark.console.core.bean.DockerConfig; +import org.apache.streampark.console.core.bean.LineageConfig; import org.apache.streampark.console.core.bean.MavenConfig; import org.apache.streampark.console.core.bean.ResponseResult; import org.apache.streampark.console.core.bean.SenderEmail; @@ -54,6 +55,12 @@ public interface SettingService extends IService { String KEY_DEFAULT_ENGINE = "engine.default"; + // lineage + String KEY_LINEAGE_GRAVITINO_ADDRESS = "lineage.gravitino.address"; + String KEY_LINEAGE_GRAVITINO_TOKEN = "lineage.gravitino.token"; + String KEY_LINEAGE_GRAVITINO_NAMESPACE = "lineage.gravitino.namespace"; + String KEY_LINEAGE_FLINK_NATIVE_LISTENER_ENABLE = "lineage.flink.native.listener.enable"; + /** * Retrieves the value of the setting associated with the specified key. * @@ -84,6 +91,13 @@ public interface SettingService extends IService { */ DockerConfig getDockerConfig(); + /** + * Retrieves the Gravitino lineage configuration settings. + * + * @return The LineageConfig object containing the Gravitino lineage configuration settings. + */ + LineageConfig getLineageConfig(); + /** * Retrieves the StreamPark address. * diff --git a/streampark-console/streampark-console-service/src/main/java/org/apache/streampark/console/core/service/application/impl/FlinkApplicationActionServiceImpl.java b/streampark-console/streampark-console-service/src/main/java/org/apache/streampark/console/core/service/application/impl/FlinkApplicationActionServiceImpl.java index f6570c4920..c66a72ec3d 100644 --- a/streampark-console/streampark-console-service/src/main/java/org/apache/streampark/console/core/service/application/impl/FlinkApplicationActionServiceImpl.java +++ b/streampark-console/streampark-console-service/src/main/java/org/apache/streampark/console/core/service/application/impl/FlinkApplicationActionServiceImpl.java @@ -37,6 +37,7 @@ import org.apache.streampark.console.base.exception.ApplicationException; import org.apache.streampark.console.base.util.Tuple2; import org.apache.streampark.console.base.util.Tuple3; +import org.apache.streampark.console.core.bean.LineageConfig; import org.apache.streampark.console.core.entity.ApplicationBuildPipeline; import org.apache.streampark.console.core.entity.ApplicationLog; import org.apache.streampark.console.core.entity.FlinkApplication; @@ -57,6 +58,7 @@ import org.apache.streampark.console.core.service.FlinkClusterService; import org.apache.streampark.console.core.service.FlinkEnvService; import org.apache.streampark.console.core.service.FlinkSqlService; +import org.apache.streampark.console.core.service.GravitinoLineageService; import org.apache.streampark.console.core.service.ResourceService; import org.apache.streampark.console.core.service.SavepointService; import org.apache.streampark.console.core.service.SettingService; @@ -81,12 +83,14 @@ import org.apache.streampark.flink.client.bean.SubmitClusterSpec; import org.apache.streampark.flink.client.bean.SubmitRequest; import org.apache.streampark.flink.client.bean.SubmitResponse; +import org.apache.streampark.flink.core.lineage.LineagePipeline; import org.apache.streampark.flink.kubernetes.FlinkK8sWatcher; import org.apache.streampark.flink.kubernetes.helper.KubernetesDeploymentHelper; import org.apache.streampark.flink.kubernetes.ingress.IngressController; import org.apache.streampark.flink.kubernetes.model.TrackId; import org.apache.streampark.flink.packer.pipeline.BuildResult; import org.apache.streampark.flink.packer.pipeline.ShadedBuildResponse; +import org.apache.streampark.flink.proxy.FlinkShimsProxy; import org.apache.commons.lang3.StringUtils; import org.apache.flink.api.common.JobID; @@ -114,7 +118,9 @@ import javax.annotation.Nonnull; import java.io.File; +import java.lang.reflect.Method; import java.net.URI; +import java.util.ArrayList; import java.util.Date; import java.util.EnumSet; import java.util.HashMap; @@ -136,6 +142,9 @@ public class FlinkApplicationActionServiceImpl implements FlinkApplicationActionService { + private static final String FLINK_SQL_LINEAGE_EXTRACTOR_CLASS = + "org.apache.streampark.flink.core.FlinkSqlLineageExtractor"; + @Qualifier("streamparkDeployExecutor") @Autowired private Executor executorService; @@ -188,6 +197,9 @@ public class FlinkApplicationActionServiceImpl @Autowired private FlinkK8sWatcherWrapper k8sWatcherWrapper; + @Autowired + private GravitinoLineageService gravitinoLineageService; + private final Map> startFutureMap = new ConcurrentHashMap<>(); @@ -424,12 +436,16 @@ public void start(FlinkApplication appParam, boolean auto) throws Exception { applicationManageService.toEffective(application); Map extraParameter = new HashMap<>(0); + final List lineagePipelines; if (application.isFlinkSql()) { FlinkSql flinkSql = flinkSqlService.getEffective(application.getId(), true); // Get the sql of the replaced placeholder String realSql = variableService.replaceVariable(application.getTeamId(), flinkSql.getSql()); flinkSql.setSql(DeflaterUtils.zipString(realSql)); extraParameter.put(ConfigKeys.KEY_FLINK_SQL(null), flinkSql.getSql()); + lineagePipelines = extractLineagePipelines(flinkEnv, application, realSql); + } else { + lineagePipelines = new ArrayList<>(); } Tuple2 userJarAndAppConf = getUserJarAndAppConf(flinkEnv, application); @@ -497,10 +513,59 @@ public void start(FlinkApplication appParam, boolean auto) throws Exception { return; } // 3) success - processForSuccess(appParam, response, applicationLog, application); + processForSuccess(appParam, response, applicationLog, application, lineagePipelines); }); } + /** + * Extracts table-level lineage from a Flink SQL job's source text via the same + * per-Flink-version, classloader-isolated mechanism {@code FlinkSqlServiceImpl.verifySql} uses + * for syntax validation — except this needs the full registered Flink Home {@code lib/} + * classpath (connector factories included), not the narrower table-planner-only one + * {@code FlinkShimsProxy.proxyVerifySql} loads, since resolving a connector-backed {@code + * CREATE TABLE} requires that connector's factory to be on the classpath. That is the same + * classpath the job itself already needs those connectors on to run, so this has no additional + * operational requirement beyond what the job already demands. + * + *

Never throws — see {@link GravitinoLineageService} for why this whole path is fail-open. + */ + List extractLineagePipelines(FlinkEnv flinkEnv, FlinkApplication application, String sql) { + if (!Boolean.TRUE.equals(application.getLineageEnable())) { + return new ArrayList<>(); + } + if (!settingService.getLineageConfig().enabled()) { + return new ArrayList<>(); + } + try { + List pipelines = FlinkShimsProxy.proxy( + flinkEnv.getFlinkVersion(), + classLoader -> { + try { + Class clazz = classLoader.loadClass(FLINK_SQL_LINEAGE_EXTRACTOR_CLASS); + Method method = clazz.getDeclaredMethod("extractLineage", String.class); + method.setAccessible(true); + Object result = method.invoke(null, sql); + if (result == null) { + return null; + } + return FlinkShimsProxy.getObject(this.getClass().getClassLoader(), result, ArrayList.class); + } catch (Exception | LinkageError e) { + // LinkageError alongside Exception: this reflective call crosses into a + // shims classloader built for a different Flink version, so a missing or + // incompatible class surfaces as NoClassDefFoundError rather than as an + // exception, and must stay as fail-open as any other lineage gap. + log.warn( + "[lineage] failed to extract lineage for application id={}", application.getId(), e); + return null; + } + }); + return pipelines == null ? new ArrayList<>() : pipelines; + } catch (Exception e) { + log.warn("[lineage] failed to extract lineage for application id={}", application.getId(), e); + return new ArrayList<>(); + } + } + @Nonnull private ApplicationLog constructAppLog(FlinkApplication application) { ApplicationLog applicationLog = new ApplicationLog(); @@ -516,7 +581,8 @@ private void processForSuccess( FlinkApplication appParam, SubmitResponse response, ApplicationLog applicationLog, - FlinkApplication flinkApplication) { + FlinkApplication flinkApplication, + List lineagePipelines) { applicationLog.setSuccess(true); if (response.flinkConfig() != null) { String jmMemory = response.flinkConfig().get(ConfigKeys.KEY_FLINK_JM_PROCESS_MEMORY()); @@ -547,10 +613,15 @@ private void processForSuccess( // if start completed, will be added task to tracking queue if (flinkApplication.isKubernetesModeJob()) { + // Kubernetes-mode jobs are tracked by a separate watcher (k8SFlinkTrackMonitor) that + // this feature does not hook into yet, so lineage START is deliberately not tracked + // here either — tracking it with no corresponding terminal hook would leak pending + // runs in gravitinoLineageService's in-memory map forever. processForK8sApp(flinkApplication, applicationLog); } else { FlinkAppHttpWatcher.setOptionState(appParam.getId(), OptionStateEnum.STARTING); FlinkAppHttpWatcher.doWatching(flinkApplication); + gravitinoLineageService.trackAndEmitStart(flinkApplication, response.jobId(), lineagePipelines); } // update app updateById(flinkApplication); @@ -793,8 +864,12 @@ private Map getProperties( properties.put(SavepointConfigOptions.SAVEPOINT_IGNORE_UNCLAIMED_STATE.key(), true); } + applyNativeLineageListenerConfig(application, properties); + Map dynamicProperties = FlinkConfigurationUtils.extractDynamicPropertiesAsJava(runtimeProperties); + // Applied last so a key the user set explicitly in Dynamic Properties always wins over + // anything this method injected above, including the native lineage listener config. properties.putAll(dynamicProperties); ResolveOrder resolveOrder = ResolveOrder.of(application.getResolveOrder()); if (resolveOrder != null) { @@ -804,6 +879,42 @@ private Map getProperties( return properties; } + /** + * Injects the official {@code openlineage-flink} job-status-changed-listener config. Covers + * what {@link #extractLineagePipelines} cannot: Custom Code (jar) Flink jobs, and — since the + * official implementation currently only produces lineage for the Kafka connector, per its own + * documented scope — Kafka-sourced tables in SQL jobs too. + * + *

Gated on two independent switches, both required: this application's own lineage switch, + * and the global "native listener" switch. The latter defaults to enabled but only actually + * injects anything once a Gravitino address is configured (see {@link + * LineageConfig#enabled()}) — an operator who has not yet placed the {@code openlineage-flink} + * jar in the registered Flink Home's {@code lib/} gets zero behavior change, not a broken job: + * without this guard, injecting {@code execution.job-status-changed-listeners} against a + * cluster missing that jar would fail every job at startup with a listener-factory + * ClassNotFoundException. + */ + void applyNativeLineageListenerConfig(FlinkApplication application, Map properties) { + if (!Boolean.TRUE.equals(application.getLineageEnable())) { + return; + } + LineageConfig lineageConfig = settingService.getLineageConfig(); + if (!lineageConfig.enabled() || !lineageConfig.isFlinkNativeListenerEnable()) { + return; + } + properties.put( + "execution.job-status-changed-listeners", + "io.openlineage.flink.listener.OpenLineageJobStatusChangedListenerFactory"); + properties.put("openlineage.transport.type", "http"); + properties.put("openlineage.transport.url", lineageConfig.getGravitinoAddress()); + properties.put("openlineage.transport.endpoint", LineageConfig.LINEAGE_ENDPOINT_PATH); + if (StringUtils.isNotBlank(lineageConfig.getGravitinoToken())) { + properties.put("openlineage.transport.auth.type", "api_key"); + properties.put("openlineage.transport.auth.apiKey", lineageConfig.getGravitinoToken()); + } + properties.put("openlineage.job.namespace", lineageConfig.namespaceOrDefault()); + } + private void doAbort(Long id) { FlinkApplication application = getById(id); application.setOptionState(OptionStateEnum.NONE.getValue()); diff --git a/streampark-console/streampark-console-service/src/main/java/org/apache/streampark/console/core/service/application/impl/FlinkApplicationManageServiceImpl.java b/streampark-console/streampark-console-service/src/main/java/org/apache/streampark/console/core/service/application/impl/FlinkApplicationManageServiceImpl.java index 5d6a489e0b..39f97e0279 100644 --- a/streampark-console/streampark-console-service/src/main/java/org/apache/streampark/console/core/service/application/impl/FlinkApplicationManageServiceImpl.java +++ b/streampark-console/streampark-console-service/src/main/java/org/apache/streampark/console/core/service/application/impl/FlinkApplicationManageServiceImpl.java @@ -400,6 +400,7 @@ public Long copy(FlinkApplication appParam) { newApp.setK8sJmPodTemplate(persist.getK8sJmPodTemplate()); newApp.setK8sTmPodTemplate(persist.getK8sTmPodTemplate()); newApp.setK8sHadoopIntegration(persist.getK8sHadoopIntegration()); + newApp.setLineageEnable(persist.getLineageEnable()); newApp.setDescription(persist.getDescription()); newApp.setAlertId(persist.getAlertId()); newApp.setCpFailureAction(persist.getCpFailureAction()); @@ -543,6 +544,7 @@ public boolean update(FlinkApplication appParam) { application.setCpFailureRateInterval(appParam.getCpFailureRateInterval()); application.setCpMaxFailureInterval(appParam.getCpMaxFailureInterval()); application.setTags(appParam.getTags()); + application.setLineageEnable(appParam.getLineageEnable()); switch (appParam.getDeployModeEnum()) { case YARN_APPLICATION: diff --git a/streampark-console/streampark-console-service/src/main/java/org/apache/streampark/console/core/service/application/impl/SparkApplicationActionServiceImpl.java b/streampark-console/streampark-console-service/src/main/java/org/apache/streampark/console/core/service/application/impl/SparkApplicationActionServiceImpl.java index 8f40380901..2a0fb9d2ce 100644 --- a/streampark-console/streampark-console-service/src/main/java/org/apache/streampark/console/core/service/application/impl/SparkApplicationActionServiceImpl.java +++ b/streampark-console/streampark-console-service/src/main/java/org/apache/streampark/console/core/service/application/impl/SparkApplicationActionServiceImpl.java @@ -31,6 +31,7 @@ import org.apache.streampark.common.util.SparkConfigurationUtils; import org.apache.streampark.console.base.exception.ApiAlertException; import org.apache.streampark.console.base.exception.ApplicationException; +import org.apache.streampark.console.core.bean.LineageConfig; import org.apache.streampark.console.core.entity.ApplicationBuildPipeline; import org.apache.streampark.console.core.entity.ApplicationLog; import org.apache.streampark.console.core.entity.Resource; @@ -46,6 +47,7 @@ import org.apache.streampark.console.core.enums.SparkOptionStateEnum; import org.apache.streampark.console.core.mapper.SparkApplicationMapper; import org.apache.streampark.console.core.service.ResourceService; +import org.apache.streampark.console.core.service.SettingService; import org.apache.streampark.console.core.service.SparkEnvService; import org.apache.streampark.console.core.service.SparkSqlService; import org.apache.streampark.console.core.service.VariableService; @@ -84,6 +86,7 @@ import java.util.Date; import java.util.EnumSet; import java.util.HashMap; +import java.util.LinkedHashMap; import java.util.List; import java.util.Map; import java.util.Set; @@ -130,6 +133,9 @@ public class SparkApplicationActionServiceImpl @Autowired private ResourceService resourceService; + @Autowired + private SettingService settingService; + private final Map> startJobFutureMap = new ConcurrentHashMap<>(); private final Map> cancelJobFutureMap = new ConcurrentHashMap<>(); @@ -313,6 +319,10 @@ public void start(SparkApplication appParam, boolean auto) throws Exception { // Get the args after placeholder replacement String applicationArgs = variableService.replaceVariable(application.getTeamId(), application.getAppArgs()); + Map sparkProperties = + SparkConfigurationUtils.extractPropertiesAsJava(application.getAppProperties()); + applyLineageConfig(application, sparkProperties); + SubmitRequest submitRequest = new SubmitRequest( sparkEnv.getSparkVersion(), SparkDeployMode.of(application.getDeployMode()), @@ -322,7 +332,7 @@ public void start(SparkApplication appParam, boolean auto) throws Exception { application.getAppName(), application.getMainClass(), appConf, - SparkConfigurationUtils.extractPropertiesAsJava(application.getAppProperties()), + sparkProperties, SparkConfigurationUtils.extractArgumentsAsJava(applicationArgs), application.getApplicationType(), application.getHadoopUser(), @@ -409,6 +419,38 @@ private void starting(SparkApplication application) { updateById(application); } + /** + * Injects the official openlineage-spark listener config so this run reports OpenLineage + * events to Gravitino. Fail-open by design: a missing Gravitino address, a disabled + * application-level switch, or any lookup failure here must never block job submission — it + * just means this run reports no lineage. Keys already present in {@code sparkProperties} (the + * user's own {@code appProperties}) are left untouched rather than overridden. + */ + void applyLineageConfig(SparkApplication application, Map sparkProperties) { + if (!Boolean.TRUE.equals(application.getLineageEnable())) { + return; + } + LineageConfig lineageConfig = settingService.getLineageConfig(); + if (!lineageConfig.enabled()) { + return; + } + Map lineageProperties = new LinkedHashMap<>(); + lineageProperties.put("spark.extraListeners", "io.openlineage.spark.agent.OpenLineageSparkListener"); + lineageProperties.put("spark.openlineage.transport.type", "http"); + lineageProperties.put("spark.openlineage.transport.url", lineageConfig.getGravitinoAddress()); + lineageProperties.put("spark.openlineage.transport.endpoint", LineageConfig.LINEAGE_ENDPOINT_PATH); + if (StringUtils.isNotBlank(lineageConfig.getGravitinoToken())) { + lineageProperties.put("spark.openlineage.transport.auth.type", "api_key"); + lineageProperties.put("spark.openlineage.transport.auth.apiKey", lineageConfig.getGravitinoToken()); + } + // Always set, defaulted the same way the Flink path defaults it: leaving it unset would let + // OpenLineage pick its own default and land Spark's datasets in a different namespace from + // Flink's, in the same StreamPark install reporting to the same Gravitino. + lineageProperties.put("spark.openlineage.namespace", lineageConfig.namespaceOrDefault()); + lineageProperties.put("spark.openlineage.columnLineage.datasetLineageEnabled", "true"); + lineageProperties.forEach(sparkProperties::putIfAbsent); + } + private Tuple2 getUserJarAndAppConf( SparkEnv sparkEnv, SparkApplication application) { SparkDeployMode deployModeEnum = application.getDeployModeEnum(); diff --git a/streampark-console/streampark-console-service/src/main/java/org/apache/streampark/console/core/service/application/impl/SparkApplicationManageServiceImpl.java b/streampark-console/streampark-console-service/src/main/java/org/apache/streampark/console/core/service/application/impl/SparkApplicationManageServiceImpl.java index b0cf67c829..89a4f22c31 100644 --- a/streampark-console/streampark-console-service/src/main/java/org/apache/streampark/console/core/service/application/impl/SparkApplicationManageServiceImpl.java +++ b/streampark-console/streampark-console-service/src/main/java/org/apache/streampark/console/core/service/application/impl/SparkApplicationManageServiceImpl.java @@ -334,6 +334,7 @@ public Long copy(SparkApplication appParam) { newApp.setK8sHadoopIntegration(oldApp.getK8sHadoopIntegration()); newApp.setHadoopUser(oldApp.getHadoopUser()); + newApp.setLineageEnable(oldApp.getLineageEnable()); newApp.setRestartSize(oldApp.getRestartSize()); newApp.setState(SparkAppStateEnum.ADDED.getValue()); newApp.setOptions(oldApp.getOptions()); @@ -442,6 +443,7 @@ public boolean update(SparkApplication appParam) { application.setAlertId(appParam.getAlertId()); application.setRestartSize(appParam.getRestartSize()); application.setTags(appParam.getTags()); + application.setLineageEnable(appParam.getLineageEnable()); switch (appParam.getDeployModeEnum()) { case YARN_CLUSTER: diff --git a/streampark-console/streampark-console-service/src/main/java/org/apache/streampark/console/core/service/impl/GravitinoLineageServiceImpl.java b/streampark-console/streampark-console-service/src/main/java/org/apache/streampark/console/core/service/impl/GravitinoLineageServiceImpl.java new file mode 100644 index 0000000000..d8ff5111a8 --- /dev/null +++ b/streampark-console/streampark-console-service/src/main/java/org/apache/streampark/console/core/service/impl/GravitinoLineageServiceImpl.java @@ -0,0 +1,248 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one or more + * contributor license agreements. See the NOTICE file distributed with + * this work for additional information regarding copyright ownership. + * The ASF licenses this file to You under the Apache License, Version 2.0 + * (the "License"); you may not use this file except in compliance with + * the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +package org.apache.streampark.console.core.service.impl; + +import org.apache.streampark.console.core.bean.LineageConfig; +import org.apache.streampark.console.core.entity.FlinkApplication; +import org.apache.streampark.console.core.service.GravitinoLineageService; +import org.apache.streampark.console.core.service.SettingService; +import org.apache.streampark.flink.core.lineage.LineageDataset; +import org.apache.streampark.flink.core.lineage.LineagePipeline; + +import org.apache.commons.lang3.StringUtils; + +import com.github.benmanes.caffeine.cache.Caffeine; +import io.openlineage.client.OpenLineage; +import io.openlineage.client.OpenLineage.RunEvent.EventType; +import io.openlineage.client.OpenLineageClient; +import io.openlineage.client.transports.ApiKeyTokenProvider; +import io.openlineage.client.transports.HttpConfig; +import io.openlineage.client.transports.HttpTransport; +import lombok.extern.slf4j.Slf4j; +import org.springframework.beans.factory.annotation.Autowired; +import org.springframework.stereotype.Service; + +import java.net.URI; +import java.nio.charset.StandardCharsets; +import java.time.Duration; +import java.time.ZoneOffset; +import java.time.ZonedDateTime; +import java.util.ArrayList; +import java.util.Collections; +import java.util.List; +import java.util.Map; +import java.util.UUID; +import java.util.concurrent.ConcurrentMap; + +@Slf4j +@Service +public class GravitinoLineageServiceImpl implements GravitinoLineageService { + + /** + * Same run-facet path Gravitino's {@code JdbcLineageStorage.runningAppId(...)} already parses + * on the receiving end (field name is engine-agnostic on the wire despite the "spark" prefix — + * that convention predates this Flink integration and is kept here for cross-emitter + * consistency, not reinvented). + */ + private static final String RUN_FACET_APP_ID_KEY = "spark_properties"; + + private static final String RUN_FACET_APP_ID_PROPERTY = "spark.app.id"; + + private static final URI PRODUCER = URI.create("https://streampark.apache.org/"); + + /** + * How long a started run stays eligible for its terminal event. A run whose application is + * deleted, or whose terminal state is never observed (Console restarted, watcher stopped + * tracking it), would otherwise pin its entry forever — this map lives for the whole Console + * process, so an unbounded one is a slow leak. The bound is time, not size: a legitimate + * streaming job may run for weeks before its COMPLETE, and evicting it because newer jobs + * started would lose the terminal event for the longest-running jobs first, which is exactly + * backwards. + */ + private static final Duration PENDING_RUN_TTL = Duration.ofDays(30); + + @Autowired + private SettingService settingService; + + /** In-memory only — see class contract in {@link GravitinoLineageService}. */ + private final ConcurrentMap pendingRuns = + Caffeine.newBuilder().expireAfterWrite(PENDING_RUN_TTL).build().asMap(); + + @Override + public void trackAndEmitStart( + FlinkApplication application, String flinkJobIdHex, + List pipelines) { + if (pipelines == null || pipelines.isEmpty()) { + return; + } + LineageConfig config; + try { + config = settingService.getLineageConfig(); + } catch (Exception e) { + log.warn("[lineage] failed to read lineage config, skipping START for application id={}", + application.getId(), e); + return; + } + if (!config.enabled()) { + return; + } + PendingRun run = + new PendingRun(flinkJobIdHex, config.namespaceOrDefault(), application.getJobName(), pipelines); + emitRunEvents(config, EventType.START, application.getId(), run); + // Tracked regardless of individual emit failures above: a later terminal call is itself + // independently fail-open (see emitTerminal), so there is no harm in attempting it even for + // a pipeline whose START never reached Gravitino — only a missed opportunity to close out + // the ones that did. + pendingRuns.put(application.getId(), run); + } + + @Override + public void emitTerminal(Long appId, boolean success) { + PendingRun run = pendingRuns.remove(appId); + if (run == null) { + return; + } + LineageConfig config; + try { + config = settingService.getLineageConfig(); + } catch (Exception e) { + log.warn("[lineage] failed to read lineage config, skipping terminal event for application id={}", appId, + e); + return; + } + if (!config.enabled()) { + return; + } + emitRunEvents(config, success ? EventType.COMPLETE : EventType.FAIL, appId, run); + } + + /** + * Emits one event per pipeline of a run over a single client. Fail-open at both levels: an + * unusable client costs the run its events, but one pipeline's failed emit must not cost the + * remaining pipelines of the same job theirs. + */ + private void emitRunEvents(LineageConfig config, EventType eventType, Long appId, PendingRun run) { + try (OpenLineageClient client = buildClient(config)) { + OpenLineage openLineage = new OpenLineage(PRODUCER); + for (LineagePipeline pipeline : run.pipelines) { + emitOne(client, openLineage, eventType, appId, run, pipeline); + } + } catch (Exception e) { + log.warn("[lineage] failed to build Gravitino client for application id={}", appId, e); + } + } + + private void emitOne( + OpenLineageClient client, + OpenLineage openLineage, + EventType eventType, + Long appId, + PendingRun run, + LineagePipeline pipeline) { + try { + UUID runId = runIdFor(run.jobIdHex, pipeline.output()); + // The Flink JobID travels as a run facet on START only — a terminal event is matched to + // its run by runId, and re-sending the facet would only restate what START established. + String startFacetJobIdHex = eventType == EventType.START ? run.jobIdHex : null; + client.emit( + buildEvent( + openLineage, eventType, runId, run.jobNamespace, run.jobName, startFacetJobIdHex, pipeline)); + } catch (Exception e) { + log.warn( + "[lineage] failed to emit {} for application id={}, sink={}", + eventType, + appId, + pipeline.output(), + e); + } + } + + /** + * Deterministic OpenLineage runId for one (Flink JobID, sink dataset) pair, stable across a + * pipeline's START/COMPLETE/FAIL. Must stay byte-identical to the same algorithm used elsewhere + * against this Gravitino deployment — this is what lets independently-emitted events for the + * same run agree on its identity without any shared state. + */ + static UUID runIdFor(String flinkJobIdHex, LineageDataset output) { + String key = "flink-job:" + flinkJobIdHex + ":" + output.namespace() + "/" + output.name(); + return UUID.nameUUIDFromBytes(key.getBytes(StandardCharsets.UTF_8)); + } + + /** Test-only observation hook into the in-memory pending-run tracking. */ + boolean hasPendingRun(Long appId) { + return pendingRuns.containsKey(appId); + } + + private OpenLineage.RunEvent buildEvent( + OpenLineage openLineage, + EventType eventType, + UUID runId, + String jobNamespace, + String jobName, + String startFacetJobIdHex, + LineagePipeline pipeline) { + OpenLineage.RunFacetsBuilder facetsBuilder = openLineage.newRunFacetsBuilder(); + if (startFacetJobIdHex != null) { + OpenLineage.DefaultRunFacet appIdFacet = new OpenLineage.DefaultRunFacet(PRODUCER); + appIdFacet + .getAdditionalProperties() + .put("properties", Map.of(RUN_FACET_APP_ID_PROPERTY, startFacetJobIdHex)); + facetsBuilder.put(RUN_FACET_APP_ID_KEY, appIdFacet); + } + OpenLineage.Run run = openLineage.newRun(runId, facetsBuilder.build()); + OpenLineage.Job job = openLineage.newJob(jobNamespace, jobName, openLineage.newJobFacetsBuilder().build()); + + List inputs = new ArrayList<>(); + for (LineageDataset input : pipeline.inputs()) { + inputs.add(openLineage.newInputDataset(input.namespace(), input.name(), null, null)); + } + List outputs = + Collections.singletonList( + openLineage.newOutputDataset( + pipeline.output().namespace(), pipeline.output().name(), null, null)); + + return openLineage.newRunEvent(ZonedDateTime.now(ZoneOffset.UTC), eventType, run, job, inputs, outputs); + } + + private OpenLineageClient buildClient(LineageConfig config) { + HttpConfig httpConfig = new HttpConfig(); + httpConfig.setUrl(URI.create(config.getGravitinoAddress())); + httpConfig.setEndpoint(LineageConfig.LINEAGE_ENDPOINT_PATH); + if (StringUtils.isNotBlank(config.getGravitinoToken())) { + ApiKeyTokenProvider tokenProvider = new ApiKeyTokenProvider(); + tokenProvider.setApiKey(config.getGravitinoToken()); + httpConfig.setAuth(tokenProvider); + } + return OpenLineageClient.builder().transport(new HttpTransport(httpConfig)).build(); + } + + private static final class PendingRun { + + private final String jobIdHex; + private final String jobNamespace; + private final String jobName; + private final List pipelines; + + private PendingRun(String jobIdHex, String jobNamespace, String jobName, List pipelines) { + this.jobIdHex = jobIdHex; + this.jobNamespace = jobNamespace; + this.jobName = jobName; + this.pipelines = pipelines; + } + } +} diff --git a/streampark-console/streampark-console-service/src/main/java/org/apache/streampark/console/core/service/impl/SettingServiceImpl.java b/streampark-console/streampark-console-service/src/main/java/org/apache/streampark/console/core/service/impl/SettingServiceImpl.java index dcc9d9f9d9..dca2f35e28 100644 --- a/streampark-console/streampark-console-service/src/main/java/org/apache/streampark/console/core/service/impl/SettingServiceImpl.java +++ b/streampark-console/streampark-console-service/src/main/java/org/apache/streampark/console/core/service/impl/SettingServiceImpl.java @@ -18,6 +18,7 @@ package org.apache.streampark.console.core.service.impl; import org.apache.streampark.console.core.bean.DockerConfig; +import org.apache.streampark.console.core.bean.LineageConfig; import org.apache.streampark.console.core.bean.MavenConfig; import org.apache.streampark.console.core.bean.ResponseResult; import org.apache.streampark.console.core.bean.SenderEmail; @@ -124,6 +125,23 @@ public DockerConfig getDockerConfig() { return dockerConfig; } + @Override + public LineageConfig getLineageConfig() { + LineageConfig lineageConfig = new LineageConfig(); + lineageConfig.setGravitinoAddress( + SETTINGS.getOrDefault(SettingService.KEY_LINEAGE_GRAVITINO_ADDRESS, emptySetting).getSettingValue()); + lineageConfig.setGravitinoToken( + SETTINGS.getOrDefault(SettingService.KEY_LINEAGE_GRAVITINO_TOKEN, emptySetting).getSettingValue()); + lineageConfig.setGravitinoNamespace( + SETTINGS.getOrDefault(SettingService.KEY_LINEAGE_GRAVITINO_NAMESPACE, emptySetting).getSettingValue()); + lineageConfig.setFlinkNativeListenerEnable( + Boolean.parseBoolean( + SETTINGS + .getOrDefault(SettingService.KEY_LINEAGE_FLINK_NATIVE_LISTENER_ENABLE, emptySetting) + .getSettingValue())); + return lineageConfig; + } + @Override public String getStreamParkAddress() { return SETTINGS diff --git a/streampark-console/streampark-console-service/src/main/java/org/apache/streampark/console/core/watcher/FlinkAppHttpWatcher.java b/streampark-console/streampark-console-service/src/main/java/org/apache/streampark/console/core/watcher/FlinkAppHttpWatcher.java index 845b215e8b..c521d5d573 100755 --- a/streampark-console/streampark-console-service/src/main/java/org/apache/streampark/console/core/watcher/FlinkAppHttpWatcher.java +++ b/streampark-console/streampark-console-service/src/main/java/org/apache/streampark/console/core/watcher/FlinkAppHttpWatcher.java @@ -36,6 +36,7 @@ import org.apache.streampark.console.core.metrics.flink.Overview; import org.apache.streampark.console.core.metrics.yarn.YarnAppInfo; import org.apache.streampark.console.core.service.FlinkClusterService; +import org.apache.streampark.console.core.service.GravitinoLineageService; import org.apache.streampark.console.core.service.SavepointService; import org.apache.streampark.console.core.service.alert.AlertService; import org.apache.streampark.console.core.service.application.FlinkApplicationActionService; @@ -99,6 +100,9 @@ public class FlinkAppHttpWatcher { @Autowired private SavepointService savepointService; + @Autowired + private GravitinoLineageService gravitinoLineageService; + // track interval every 5 seconds public static final Duration WATCHING_INTERVAL = Duration.ofSeconds(5); @@ -441,7 +445,8 @@ private void handleRunningState( private void doPersistMetrics(FlinkApplication application, boolean stopWatch) { Long appId = application.getId(); - if (FlinkAppStateEnum.isEndState(application.getState())) { + boolean isEndState = FlinkAppStateEnum.isEndState(application.getState()); + if (isEndState) { application.setOverview(null); application.setTotalTM(null); application.setTotalSlot(null); @@ -461,9 +466,18 @@ private void doPersistMetrics(FlinkApplication application, boolean stopWatch) { if (!nowEvent.equals(event)) { PREVIOUS_STATUS.put(appId, nowEvent); applicationManageService.persistMetrics(application); + if (isEndState) { + gravitinoLineageService.emitTerminal(appId, isLineageSuccessState(application.getState())); + } } } + /** FINISHED/SUCCEEDED report OpenLineage COMPLETE; every other end state reports FAIL. */ + boolean isLineageSuccessState(Integer state) { + FlinkAppStateEnum flinkAppState = FlinkAppStateEnum.getState(state); + return flinkAppState == FlinkAppStateEnum.FINISHED || flinkAppState == FlinkAppStateEnum.SUCCEEDED; + } + /** * Handle not running task * diff --git a/streampark-console/streampark-console-service/src/main/resources/db/data-h2.sql b/streampark-console/streampark-console-service/src/main/resources/db/data-h2.sql index 7dac02fb6b..952f58f9fe 100644 --- a/streampark-console/streampark-console-service/src/main/resources/db/data-h2.sql +++ b/streampark-console/streampark-console-service/src/main/resources/db/data-h2.sql @@ -326,6 +326,10 @@ insert into `t_setting` values (12, 'docker.register.user', null, 'Docker Regist insert into `t_setting` values (13, 'docker.register.password', null, 'Docker Register Password', 'Docker container service authentication password', 1); insert into `t_setting` values (14, 'docker.register.namespace', null, 'Docker namespace', 'Namespace for docker image used in docker building env and target image register', 1); insert into `t_setting` values (15, 'ingress.mode.default', null, 'Ingress domain address', 'Automatically generate an nginx-based ingress by passing in a domain name', 1); +insert into `t_setting` values (16, 'lineage.gravitino.address', null, 'Gravitino Address', 'Base URL of the Gravitino server lineage events are reported to, e.g. http://host:8090', 1); +insert into `t_setting` values (17, 'lineage.gravitino.token', null, 'Gravitino Auth Token', 'Bearer token forwarded to Gravitino, required once Gravitino oauth authentication is enabled', 1); +insert into `t_setting` values (18, 'lineage.gravitino.namespace', null, 'Gravitino Lineage Namespace', 'OpenLineage job/dataset namespace StreamPark reports under', 1); +insert into `t_setting` values (19, 'lineage.flink.native.listener.enable', 'true', 'Enable Flink Native OpenLineage Listener', 'Whether to also inject the official openlineage-flink job-status-changed-listener config; only takes effect once Gravitino Address is set', 2); -- ---------------------------- -- Records of t_user diff --git a/streampark-console/streampark-console-service/src/main/resources/db/schema-h2.sql b/streampark-console/streampark-console-service/src/main/resources/db/schema-h2.sql index 3197caf56c..c45dbad761 100644 --- a/streampark-console/streampark-console-service/src/main/resources/db/schema-h2.sql +++ b/streampark-console/streampark-console-service/src/main/resources/db/schema-h2.sql @@ -91,6 +91,7 @@ create table if not exists `t_flink_app` ( `default_mode_ingress` text , `tags` varchar(500) default null, `hadoop_user` varchar(500) default null, + `lineage_enable` tinyint default 0, primary key(`id`) ); @@ -564,6 +565,7 @@ create table if not exists `t_spark_app` ( `k8s_executor_pod_template` text, `k8s_hadoop_integration` tinyint default 0, `hadoop_user` varchar(64) default null, + `lineage_enable` tinyint default 0, `restart_size` int default null, `restart_count` int default null, `state` int default null, diff --git a/streampark-console/streampark-console-service/src/test/java/org/apache/streampark/console/core/service/application/impl/FlinkApplicationActionServiceImplTest.java b/streampark-console/streampark-console-service/src/test/java/org/apache/streampark/console/core/service/application/impl/FlinkApplicationActionServiceImplTest.java new file mode 100644 index 0000000000..165fb65532 --- /dev/null +++ b/streampark-console/streampark-console-service/src/test/java/org/apache/streampark/console/core/service/application/impl/FlinkApplicationActionServiceImplTest.java @@ -0,0 +1,157 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one or more + * contributor license agreements. See the NOTICE file distributed with + * this work for additional information regarding copyright ownership. + * The ASF licenses this file to You under the Apache License, Version 2.0 + * (the "License"); you may not use this file except in compliance with + * the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +package org.apache.streampark.console.core.service.application.impl; + +import org.apache.streampark.console.core.bean.LineageConfig; +import org.apache.streampark.console.core.entity.FlinkApplication; +import org.apache.streampark.console.core.service.SettingService; +import org.apache.streampark.flink.core.lineage.LineagePipeline; + +import org.junit.jupiter.api.Test; +import org.junit.jupiter.api.extension.ExtendWith; +import org.mockito.InjectMocks; +import org.mockito.Mock; +import org.mockito.junit.jupiter.MockitoExtension; +import org.mockito.junit.jupiter.MockitoSettings; +import org.mockito.quality.Strictness; + +import java.util.HashMap; +import java.util.List; +import java.util.Map; + +import static org.assertj.core.api.Assertions.assertThat; +import static org.mockito.Mockito.lenient; +import static org.mockito.Mockito.verifyNoInteractions; + +@ExtendWith(MockitoExtension.class) +@MockitoSettings(strictness = Strictness.LENIENT) +class FlinkApplicationActionServiceImplTest { + + @Mock + private SettingService settingService; + + @InjectMocks + private FlinkApplicationActionServiceImpl service; + + private static LineageConfig enabledConfig(boolean nativeListenerEnable) { + LineageConfig config = new LineageConfig(); + config.setGravitinoAddress("http://192.168.10.132:8090"); + config.setGravitinoToken("test-token"); + config.setGravitinoNamespace("streampark"); + config.setFlinkNativeListenerEnable(nativeListenerEnable); + return config; + } + + @Test + void applyNativeLineageListenerConfigDoesNothingWhenAppSwitchOff() { + FlinkApplication application = new FlinkApplication(); + application.setLineageEnable(false); + Map properties = new HashMap<>(); + + service.applyNativeLineageListenerConfig(application, properties); + + verifyNoInteractions(settingService); + assertThat(properties).isEmpty(); + } + + @Test + void applyNativeLineageListenerConfigDoesNothingWhenGravitinoAddressUnset() { + FlinkApplication application = new FlinkApplication(); + application.setLineageEnable(true); + lenient().when(settingService.getLineageConfig()).thenReturn(new LineageConfig()); + Map properties = new HashMap<>(); + + service.applyNativeLineageListenerConfig(application, properties); + + assertThat(properties).isEmpty(); + } + + @Test + void applyNativeLineageListenerConfigDoesNothingWhenGlobalListenerSwitchOff() { + FlinkApplication application = new FlinkApplication(); + application.setLineageEnable(true); + lenient().when(settingService.getLineageConfig()).thenReturn(enabledConfig(false)); + Map properties = new HashMap<>(); + + service.applyNativeLineageListenerConfig(application, properties); + + assertThat(properties).isEmpty(); + } + + @Test + void applyNativeLineageListenerConfigInjectsWhenBothSwitchesOn() { + FlinkApplication application = new FlinkApplication(); + application.setLineageEnable(true); + lenient().when(settingService.getLineageConfig()).thenReturn(enabledConfig(true)); + Map properties = new HashMap<>(); + + service.applyNativeLineageListenerConfig(application, properties); + + assertThat(properties) + .containsEntry( + "execution.job-status-changed-listeners", + "io.openlineage.flink.listener.OpenLineageJobStatusChangedListenerFactory") + .containsEntry("openlineage.transport.type", "http") + .containsEntry("openlineage.transport.url", "http://192.168.10.132:8090") + .containsEntry("openlineage.transport.endpoint", "/api/lineage") + .containsEntry("openlineage.transport.auth.type", "api_key") + .containsEntry("openlineage.transport.auth.apiKey", "test-token") + .containsEntry("openlineage.job.namespace", "streampark"); + } + + @Test + void applyNativeLineageListenerConfigDoesNotOverrideUserSuppliedProperties() { + FlinkApplication application = new FlinkApplication(); + application.setLineageEnable(true); + lenient().when(settingService.getLineageConfig()).thenReturn(enabledConfig(true)); + Map properties = new HashMap<>(); + properties.put("execution.job-status-changed-listeners", "com.example.MyOwnListenerFactory"); + + service.applyNativeLineageListenerConfig(application, properties); + + // this method itself is a plain put — the "don't override the user" contract is enforced + // by the caller (getProperties) applying Dynamic Properties after this method runs, not by + // this method checking for an existing value; this test documents that division of duty. + assertThat(properties) + .containsEntry( + "execution.job-status-changed-listeners", + "io.openlineage.flink.listener.OpenLineageJobStatusChangedListenerFactory"); + } + + @Test + void extractLineagePipelinesReturnsEmptyWhenAppSwitchOff() { + FlinkApplication application = new FlinkApplication(); + application.setLineageEnable(false); + + List pipelines = service.extractLineagePipelines(null, application, "SELECT 1"); + + verifyNoInteractions(settingService); + assertThat(pipelines).isEmpty(); + } + + @Test + void extractLineagePipelinesReturnsEmptyWhenGlobalConfigDisabled() { + FlinkApplication application = new FlinkApplication(); + application.setLineageEnable(true); + lenient().when(settingService.getLineageConfig()).thenReturn(new LineageConfig()); + + List pipelines = service.extractLineagePipelines(null, application, "SELECT 1"); + + assertThat(pipelines).isEmpty(); + } +} diff --git a/streampark-console/streampark-console-service/src/test/java/org/apache/streampark/console/core/service/application/impl/SparkApplicationActionServiceImplTest.java b/streampark-console/streampark-console-service/src/test/java/org/apache/streampark/console/core/service/application/impl/SparkApplicationActionServiceImplTest.java new file mode 100644 index 0000000000..0a97f8301e --- /dev/null +++ b/streampark-console/streampark-console-service/src/test/java/org/apache/streampark/console/core/service/application/impl/SparkApplicationActionServiceImplTest.java @@ -0,0 +1,123 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one or more + * contributor license agreements. See the NOTICE file distributed with + * this work for additional information regarding copyright ownership. + * The ASF licenses this file to You under the Apache License, Version 2.0 + * (the "License"); you may not use this file except in compliance with + * the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +package org.apache.streampark.console.core.service.application.impl; + +import org.apache.streampark.console.core.bean.LineageConfig; +import org.apache.streampark.console.core.entity.SparkApplication; +import org.apache.streampark.console.core.service.SettingService; + +import org.junit.jupiter.api.Test; +import org.junit.jupiter.api.extension.ExtendWith; +import org.mockito.InjectMocks; +import org.mockito.Mock; +import org.mockito.junit.jupiter.MockitoExtension; +import org.mockito.junit.jupiter.MockitoSettings; +import org.mockito.quality.Strictness; + +import java.util.HashMap; +import java.util.Map; + +import static org.assertj.core.api.Assertions.assertThat; +import static org.mockito.Mockito.lenient; + +@ExtendWith(MockitoExtension.class) +@MockitoSettings(strictness = Strictness.LENIENT) +class SparkApplicationActionServiceImplTest { + + @Mock + private SettingService settingService; + + @InjectMocks + private SparkApplicationActionServiceImpl service; + + private LineageConfig enabledLineageConfig() { + LineageConfig config = new LineageConfig(); + config.setGravitinoAddress("http://192.168.10.132:8090"); + config.setGravitinoToken("test-token"); + config.setGravitinoNamespace("streampark"); + return config; + } + + @Test + void injectsLineageConfigWhenAppSwitchAndGlobalAddressAreBothSet() { + SparkApplication application = new SparkApplication(); + application.setLineageEnable(true); + lenient().when(settingService.getLineageConfig()).thenReturn(enabledLineageConfig()); + + Map sparkProperties = new HashMap<>(); + service.applyLineageConfig(application, sparkProperties); + + assertThat(sparkProperties) + .containsEntry("spark.extraListeners", "io.openlineage.spark.agent.OpenLineageSparkListener") + .containsEntry("spark.openlineage.transport.type", "http") + .containsEntry("spark.openlineage.transport.url", "http://192.168.10.132:8090") + .containsEntry("spark.openlineage.transport.endpoint", "/api/lineage") + .containsEntry("spark.openlineage.transport.auth.type", "api_key") + .containsEntry("spark.openlineage.transport.auth.apiKey", "test-token") + .containsEntry("spark.openlineage.namespace", "streampark") + .containsEntry("spark.openlineage.columnLineage.datasetLineageEnabled", "true"); + } + + @Test + void doesNotInjectWhenAppSwitchIsOff() { + SparkApplication application = new SparkApplication(); + application.setLineageEnable(false); + + Map sparkProperties = new HashMap<>(); + service.applyLineageConfig(application, sparkProperties); + + assertThat(sparkProperties).isEmpty(); + } + + @Test + void doesNotInjectWhenAppSwitchIsNull() { + SparkApplication application = new SparkApplication(); + + Map sparkProperties = new HashMap<>(); + service.applyLineageConfig(application, sparkProperties); + + assertThat(sparkProperties).isEmpty(); + } + + @Test + void doesNotInjectWhenGlobalGravitinoAddressIsBlank() { + SparkApplication application = new SparkApplication(); + application.setLineageEnable(true); + lenient().when(settingService.getLineageConfig()).thenReturn(new LineageConfig()); + + Map sparkProperties = new HashMap<>(); + service.applyLineageConfig(application, sparkProperties); + + assertThat(sparkProperties).isEmpty(); + } + + @Test + void doesNotOverrideUserSuppliedProperties() { + SparkApplication application = new SparkApplication(); + application.setLineageEnable(true); + lenient().when(settingService.getLineageConfig()).thenReturn(enabledLineageConfig()); + + Map sparkProperties = new HashMap<>(); + sparkProperties.put("spark.extraListeners", "com.example.MyOwnListener"); + service.applyLineageConfig(application, sparkProperties); + + assertThat(sparkProperties).containsEntry("spark.extraListeners", "com.example.MyOwnListener"); + // other lineage keys the user did not set are still injected + assertThat(sparkProperties).containsEntry("spark.openlineage.transport.type", "http"); + } +} diff --git a/streampark-console/streampark-console-service/src/test/java/org/apache/streampark/console/core/service/impl/GravitinoLineageServiceImplTest.java b/streampark-console/streampark-console-service/src/test/java/org/apache/streampark/console/core/service/impl/GravitinoLineageServiceImplTest.java new file mode 100644 index 0000000000..cb83f9509e --- /dev/null +++ b/streampark-console/streampark-console-service/src/test/java/org/apache/streampark/console/core/service/impl/GravitinoLineageServiceImplTest.java @@ -0,0 +1,153 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one or more + * contributor license agreements. See the NOTICE file distributed with + * this work for additional information regarding copyright ownership. + * The ASF licenses this file to You under the Apache License, Version 2.0 + * (the "License"); you may not use this file except in compliance with + * the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +package org.apache.streampark.console.core.service.impl; + +import org.apache.streampark.console.core.bean.LineageConfig; +import org.apache.streampark.console.core.entity.FlinkApplication; +import org.apache.streampark.console.core.service.SettingService; +import org.apache.streampark.flink.core.lineage.LineageDataset; +import org.apache.streampark.flink.core.lineage.LineagePipeline; + +import org.junit.jupiter.api.Test; +import org.junit.jupiter.api.extension.ExtendWith; +import org.mockito.InjectMocks; +import org.mockito.Mock; +import org.mockito.junit.jupiter.MockitoExtension; +import org.mockito.junit.jupiter.MockitoSettings; +import org.mockito.quality.Strictness; + +import java.util.List; +import java.util.Set; + +import static org.assertj.core.api.Assertions.assertThat; +import static org.mockito.Mockito.lenient; +import static org.mockito.Mockito.verifyNoInteractions; + +@ExtendWith(MockitoExtension.class) +@MockitoSettings(strictness = Strictness.LENIENT) +class GravitinoLineageServiceImplTest { + + @Mock + private SettingService settingService; + + @InjectMocks + private GravitinoLineageServiceImpl service; + + private static LineagePipeline pipeline() { + LineageDataset output = new LineageDataset("paimon://catalog/db", "sink_table"); + LineageDataset input = new LineageDataset("mysql-cdc://host:3306", "db.source_table"); + return new LineagePipeline(output, Set.of(input)); + } + + // Unreachable on purpose — exercises the fail-open path without a real Gravitino server. + private static LineageConfig unreachableEnabledConfig() { + LineageConfig config = new LineageConfig(); + config.setGravitinoAddress("http://127.0.0.1:1"); + config.setGravitinoNamespace("streampark"); + return config; + } + + @Test + void runIdForIsDeterministicForTheSameJobAndDataset() { + LineageDataset output = new LineageDataset("paimon://catalog/db", "sink_table"); + + java.util.UUID first = GravitinoLineageServiceImpl.runIdFor("abc123", output); + java.util.UUID second = GravitinoLineageServiceImpl.runIdFor("abc123", output); + + assertThat(first).isEqualTo(second); + } + + @Test + void runIdForDiffersAcrossDifferentJobsOrDatasets() { + LineageDataset output = new LineageDataset("paimon://catalog/db", "sink_table"); + LineageDataset otherOutput = new LineageDataset("paimon://catalog/db", "other_sink"); + + java.util.UUID a = GravitinoLineageServiceImpl.runIdFor("job-1", output); + java.util.UUID b = GravitinoLineageServiceImpl.runIdFor("job-2", output); + java.util.UUID c = GravitinoLineageServiceImpl.runIdFor("job-1", otherOutput); + + assertThat(a).isNotEqualTo(b).isNotEqualTo(c); + } + + @Test + void runIdForMatchesTheNameUuidOfTheDocumentedKeyFormat() { + LineageDataset output = new LineageDataset("paimon://catalog/db", "sink_table"); + String expectedKey = "flink-job:abc123:paimon://catalog/db/sink_table"; + + java.util.UUID actual = GravitinoLineageServiceImpl.runIdFor("abc123", output); + + assertThat(actual) + .isEqualTo(java.util.UUID.nameUUIDFromBytes(expectedKey.getBytes(java.nio.charset.StandardCharsets.UTF_8))); + } + + @Test + void trackAndEmitStartNoOpsOnEmptyPipelines() { + FlinkApplication application = new FlinkApplication(); + application.setId(1L); + + service.trackAndEmitStart(application, "job-1", List.of()); + + verifyNoInteractions(settingService); + assertThat(service.hasPendingRun(1L)).isFalse(); + } + + @Test + void trackAndEmitStartNoOpsWhenLineageDisabled() { + FlinkApplication application = new FlinkApplication(); + application.setId(1L); + lenient().when(settingService.getLineageConfig()).thenReturn(new LineageConfig()); + + service.trackAndEmitStart(application, "job-1", List.of(pipeline())); + + assertThat(service.hasPendingRun(1L)).isFalse(); + } + + @Test + void trackAndEmitStartTracksThePendingRunEvenWhenGravitinoIsUnreachable() { + FlinkApplication application = new FlinkApplication(); + application.setId(1L); + application.setJobName("test-job"); + lenient().when(settingService.getLineageConfig()).thenReturn(unreachableEnabledConfig()); + + service.trackAndEmitStart(application, "job-1", List.of(pipeline())); + + assertThat(service.hasPendingRun(1L)).isTrue(); + } + + @Test + void emitTerminalNoOpsWhenNoPendingRunIsTracked() { + // Must not throw even though nothing was ever tracked for this appId. + service.emitTerminal(999L, true); + } + + @Test + void emitTerminalConsumesThePendingRunExactlyOnce() { + FlinkApplication application = new FlinkApplication(); + application.setId(1L); + application.setJobName("test-job"); + lenient().when(settingService.getLineageConfig()).thenReturn(unreachableEnabledConfig()); + service.trackAndEmitStart(application, "job-1", List.of(pipeline())); + assertThat(service.hasPendingRun(1L)).isTrue(); + + service.emitTerminal(1L, true); + + assertThat(service.hasPendingRun(1L)).isFalse(); + // second call for the same appId is a no-op, not an error + service.emitTerminal(1L, true); + } +} diff --git a/streampark-console/streampark-console-service/src/test/java/org/apache/streampark/console/core/watcher/FlinkAppHttpWatcherLineageTest.java b/streampark-console/streampark-console-service/src/test/java/org/apache/streampark/console/core/watcher/FlinkAppHttpWatcherLineageTest.java new file mode 100644 index 0000000000..bf720b8952 --- /dev/null +++ b/streampark-console/streampark-console-service/src/test/java/org/apache/streampark/console/core/watcher/FlinkAppHttpWatcherLineageTest.java @@ -0,0 +1,49 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one or more + * contributor license agreements. See the NOTICE file distributed with + * this work for additional information regarding copyright ownership. + * The ASF licenses this file to You under the Apache License, Version 2.0 + * (the "License"); you may not use this file except in compliance with + * the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +package org.apache.streampark.console.core.watcher; + +import org.apache.streampark.console.core.enums.FlinkAppStateEnum; + +import org.junit.jupiter.api.Test; + +import static org.assertj.core.api.Assertions.assertThat; + +/** + * Covers only the pure COMPLETE-vs-FAIL mapping this watcher adds for lineage terminal events — + * not the surrounding {@code doPersistMetrics} state-tracking machinery, which is exercised + * end-to-end in {@code SpringIntegrationTestBase}-backed tests elsewhere. + */ +class FlinkAppHttpWatcherLineageTest { + + private final FlinkAppHttpWatcher watcher = new FlinkAppHttpWatcher(); + + @Test + void finishedAndSucceededMapToLineageSuccess() { + assertThat(watcher.isLineageSuccessState(FlinkAppStateEnum.FINISHED.getValue())).isTrue(); + assertThat(watcher.isLineageSuccessState(FlinkAppStateEnum.SUCCEEDED.getValue())).isTrue(); + } + + @Test + void otherEndStatesMapToLineageFailure() { + assertThat(watcher.isLineageSuccessState(FlinkAppStateEnum.FAILED.getValue())).isFalse(); + assertThat(watcher.isLineageSuccessState(FlinkAppStateEnum.CANCELED.getValue())).isFalse(); + assertThat(watcher.isLineageSuccessState(FlinkAppStateEnum.KILLED.getValue())).isFalse(); + assertThat(watcher.isLineageSuccessState(FlinkAppStateEnum.LOST.getValue())).isFalse(); + assertThat(watcher.isLineageSuccessState(FlinkAppStateEnum.TERMINATED.getValue())).isFalse(); + } +} diff --git a/streampark-console/streampark-console-webapp/src/api/flink/app.type.ts b/streampark-console/streampark-console-webapp/src/api/flink/app.type.ts index df6d96132c..17c301acd1 100644 --- a/streampark-console/streampark-console-webapp/src/api/flink/app.type.ts +++ b/streampark-console/streampark-console-webapp/src/api/flink/app.type.ts @@ -139,6 +139,7 @@ export interface AppListRecord { }; streamParkJob: boolean; hadoopUser: string; + lineageEnable?: boolean; } interface AppControl { diff --git a/streampark-console/streampark-console-webapp/src/api/spark/app.type.ts b/streampark-console/streampark-console-webapp/src/api/spark/app.type.ts index 41704111b6..5366cff306 100644 --- a/streampark-console/streampark-console-webapp/src/api/spark/app.type.ts +++ b/streampark-console/streampark-console-webapp/src/api/spark/app.type.ts @@ -63,6 +63,7 @@ export interface SparkApplication { k8sServiceAccount?: number; k8sNamespace?: string; hadoopUser?: string; + lineageEnable?: boolean; restartSize?: number; restartCount?: number; state?: AppStateEnum; diff --git a/streampark-console/streampark-console-webapp/src/locales/lang/en/flink/app.ts b/streampark-console/streampark-console-webapp/src/locales/lang/en/flink/app.ts index ab0bc56b82..844ace3428 100644 --- a/streampark-console/streampark-console-webapp/src/locales/lang/en/flink/app.ts +++ b/streampark-console/streampark-console-webapp/src/locales/lang/en/flink/app.ts @@ -69,6 +69,9 @@ export default { startTime: 'Start Time', endTime: 'End Time', hadoopUser: 'Hadoop User', + lineageEnable: 'Data Lineage', + lineageEnableTip: + "Report this job's table-level lineage to Gravitino; takes effect only once Gravitino Address is configured in System Setting", restoreModeTip: 'restore mode is supported since flink 1.15, usually, you do not have to set this parameter', release: { diff --git a/streampark-console/streampark-console-webapp/src/locales/lang/en/setting/system.ts b/streampark-console/streampark-console-webapp/src/locales/lang/en/setting/system.ts index f4b156e3e4..dacd5f8444 100644 --- a/streampark-console/streampark-console-webapp/src/locales/lang/en/setting/system.ts +++ b/streampark-console/streampark-console-webapp/src/locales/lang/en/setting/system.ts @@ -32,6 +32,9 @@ export default { ingressSetting: { name: 'Ingress Setting', }, + lineageSetting: { + name: 'Gravitino Lineage Setting', + }, }, update: { success: 'Setting updated successfully', @@ -99,4 +102,22 @@ export default { desc: 'Whether to enable SSL in the mailbox that sends the alert', }, }, + lineage: { + address: { + label: 'Gravitino Address', + desc: 'Base URL of the Gravitino server lineage events are reported to, e.g. http://host:8090', + }, + token: { + label: 'Gravitino Auth Token', + desc: 'Bearer token forwarded to Gravitino, required once Gravitino oauth authentication is enabled', + }, + namespace: { + label: 'Gravitino Lineage Namespace', + desc: 'OpenLineage job/dataset namespace StreamPark reports under', + }, + nativeListenerEnable: { + label: 'Enable Flink Native Lineage Listener', + desc: 'Whether to also inject the official openlineage-flink listener config; only takes effect once Gravitino Address is set', + }, + }, }; diff --git a/streampark-console/streampark-console-webapp/src/locales/lang/en/spark/app.ts b/streampark-console/streampark-console-webapp/src/locales/lang/en/spark/app.ts index ecc336b8f8..c95d4bc5c7 100644 --- a/streampark-console/streampark-console-webapp/src/locales/lang/en/spark/app.ts +++ b/streampark-console/streampark-console-webapp/src/locales/lang/en/spark/app.ts @@ -102,6 +102,9 @@ export default { startTime: 'Start Time', endTime: 'End Time', hadoopUser: 'Hadoop User', + lineageEnable: 'Data Lineage', + lineageEnableTip: + "Report this job's lineage to Gravitino; takes effect only once Gravitino Address is configured in System Setting", success: 'Submission Successful', appidCheck: 'appId cannot be empty!', release: { diff --git a/streampark-console/streampark-console-webapp/src/locales/lang/zh-CN/flink/app.ts b/streampark-console/streampark-console-webapp/src/locales/lang/zh-CN/flink/app.ts index 11a8b417a7..c2df6efe00 100644 --- a/streampark-console/streampark-console-webapp/src/locales/lang/zh-CN/flink/app.ts +++ b/streampark-console/streampark-console-webapp/src/locales/lang/zh-CN/flink/app.ts @@ -68,6 +68,8 @@ export default { startTime: '启动时间', endTime: '结束时间', hadoopUser: 'Hadoop User', + lineageEnable: '数据血缘', + lineageEnableTip: '上报该作业的表级血缘到 Gravitino;仅在系统设置中配置了 Gravitino 地址后生效', restoreModeTip: 'flink 1.15开始支持restore模式,一般情况下不用设置该参数', release: { releaseTitle: '该作业正在启动中.', diff --git a/streampark-console/streampark-console-webapp/src/locales/lang/zh-CN/setting/system.ts b/streampark-console/streampark-console-webapp/src/locales/lang/zh-CN/setting/system.ts index 9738766a0c..1083987cfe 100644 --- a/streampark-console/streampark-console-webapp/src/locales/lang/zh-CN/setting/system.ts +++ b/streampark-console/streampark-console-webapp/src/locales/lang/zh-CN/setting/system.ts @@ -32,6 +32,9 @@ export default { ingressSetting: { name: 'Kubernetes Ingress 配置', }, + lineageSetting: { + name: 'Gravitino 数据血缘配置', + }, }, update: { success: '设置更新成功!', @@ -99,4 +102,22 @@ export default { desc: '是否在发送告警邮箱中启用 SSL', }, }, + lineage: { + address: { + label: 'Gravitino 地址', + desc: '接收数据血缘事件的 Gravitino 服务地址,例如 http://host:8090', + }, + token: { + label: 'Gravitino 认证 Token', + desc: '透传给 Gravitino 的 Bearer token,Gravitino 开启 oauth 鉴权后必填', + }, + namespace: { + label: 'Gravitino 血缘命名空间', + desc: 'StreamPark 上报血缘事件使用的 OpenLineage job/dataset 命名空间', + }, + nativeListenerEnable: { + label: '启用 Flink 官方血缘监听器', + desc: '是否同时注入官方 openlineage-flink 监听器配置;仅在已配置 Gravitino 地址时生效', + }, + }, }; diff --git a/streampark-console/streampark-console-webapp/src/locales/lang/zh-CN/spark/app.ts b/streampark-console/streampark-console-webapp/src/locales/lang/zh-CN/spark/app.ts index e0a445901b..68b2d0148e 100644 --- a/streampark-console/streampark-console-webapp/src/locales/lang/zh-CN/spark/app.ts +++ b/streampark-console/streampark-console-webapp/src/locales/lang/zh-CN/spark/app.ts @@ -102,6 +102,8 @@ export default { startTime: '启动时间', endTime: '结束时间', hadoopUser: 'Hadoop User', + lineageEnable: '数据血缘', + lineageEnableTip: '上报该作业的血缘到 Gravitino;仅在系统设置中配置了 Gravitino 地址后生效', success: '提交成功', appidCheck: 'appId 不能为空!', release: { diff --git a/streampark-console/streampark-console-webapp/src/views/flink/app/EditFlink.vue b/streampark-console/streampark-console-webapp/src/views/flink/app/EditFlink.vue index 24aaf5c381..978fb393cc 100644 --- a/streampark-console/streampark-console-webapp/src/views/flink/app/EditFlink.vue +++ b/streampark-console/streampark-console-webapp/src/views/flink/app/EditFlink.vue @@ -89,6 +89,7 @@ jar: app.jar, description: app.description, hadoopUser: app.hadoopUser, + lineageEnable: app.lineageEnable ?? false, dynamicProperties: app.dynamicProperties, resolveOrder: app.resolveOrder, deployMode: app.deployMode, diff --git a/streampark-console/streampark-console-webapp/src/views/flink/app/EditStreamPark.vue b/streampark-console/streampark-console-webapp/src/views/flink/app/EditStreamPark.vue index 2cc118ceb6..7d1876a406 100644 --- a/streampark-console/streampark-console-webapp/src/views/flink/app/EditStreamPark.vue +++ b/streampark-console/streampark-console-webapp/src/views/flink/app/EditStreamPark.vue @@ -108,6 +108,7 @@ tags: app.tags, args: app.args || '', description: app.description, + lineageEnable: app.lineageEnable ?? false, dynamicProperties: app.dynamicProperties, resolveOrder: app.resolveOrder, versionId: app.versionId || null, diff --git a/streampark-console/streampark-console-webapp/src/views/flink/app/hooks/useCreateAndEditSchema.ts b/streampark-console/streampark-console-webapp/src/views/flink/app/hooks/useCreateAndEditSchema.ts index 62f20eab5f..b5a748b35a 100644 --- a/streampark-console/streampark-console-webapp/src/views/flink/app/hooks/useCreateAndEditSchema.ts +++ b/streampark-console/streampark-console-webapp/src/views/flink/app/hooks/useCreateAndEditSchema.ts @@ -484,6 +484,14 @@ export const useCreateAndEditSchema = ( label: t('flink.app.hadoopUser'), component: 'Input', }, + { + field: 'lineageEnable', + label: t('flink.app.lineageEnable'), + component: 'Switch', + defaultValue: false, + componentProps: { checkedChildren: 'ON', unCheckedChildren: 'OFF' }, + afterItem: () => h('span', { class: 'pop-tip' }, t('flink.app.lineageEnableTip')), + }, { field: 'description', label: t('common.description'), diff --git a/streampark-console/streampark-console-webapp/src/views/flink/app/utils/index.ts b/streampark-console/streampark-console-webapp/src/views/flink/app/utils/index.ts index b86aeef783..8f2f215738 100644 --- a/streampark-console/streampark-console-webapp/src/views/flink/app/utils/index.ts +++ b/streampark-console/streampark-console-webapp/src/views/flink/app/utils/index.ts @@ -280,6 +280,7 @@ export function handleSubmitParams( k8sRestExposedType: values.k8sRestExposedType, restartSize: values.restartSize, alertId: values.alertId, + lineageEnable: values.lineageEnable ?? false, description: values.description, k8sNamespace: values.k8sNamespace || null, clusterId: values.clusterId || null, diff --git a/streampark-console/streampark-console-webapp/src/views/setting/system/SettingList.vue b/streampark-console/streampark-console-webapp/src/views/setting/system/SettingList.vue index 1d67ccf7cd..ccbd19d5a1 100644 --- a/streampark-console/streampark-console-webapp/src/views/setting/system/SettingList.vue +++ b/streampark-console/streampark-console-webapp/src/views/setting/system/SettingList.vue @@ -40,6 +40,10 @@ 'docker.register.address': 'docker', 'alert.email.from': 'mail', 'ingress.mode.default': 'nginx', + 'lineage.gravitino.address': 'net', + 'lineage.gravitino.token': 'keys', + 'lineage.gravitino.namespace': 'namespace', + 'lineage.flink.native.listener.enable': 'connector', }; const settingTitles = { @@ -50,6 +54,10 @@ 'docker.register.address': t('setting.system.title.docker'), 'alert.email.from': t('setting.system.title.email'), 'ingress.mode.default': t('setting.system.title.ingress'), + 'lineage.gravitino.address': t('setting.system.lineage.address.label'), + 'lineage.gravitino.token': t('setting.system.lineage.token.label'), + 'lineage.gravitino.namespace': t('setting.system.lineage.namespace.label'), + 'lineage.flink.native.listener.enable': t('setting.system.lineage.nativeListenerEnable.label'), }; const settingDesc = { @@ -60,6 +68,10 @@ 'docker.register.address': t('setting.system.desc.docker'), 'alert.email.from': t('setting.system.desc.email'), 'ingress.mode.default': t('setting.system.desc.ingress'), + 'lineage.gravitino.address': t('setting.system.lineage.address.desc'), + 'lineage.gravitino.token': t('setting.system.lineage.token.desc'), + 'lineage.gravitino.namespace': t('setting.system.lineage.namespace.desc'), + 'lineage.flink.native.listener.enable': t('setting.system.lineage.nativeListenerEnable.desc'), }; const ListItem = List.Item; diff --git a/streampark-console/streampark-console-webapp/src/views/setting/system/View.vue b/streampark-console/streampark-console-webapp/src/views/setting/system/View.vue index aedb69515e..963c039761 100644 --- a/streampark-console/streampark-console-webapp/src/views/setting/system/View.vue +++ b/streampark-console/streampark-console-webapp/src/views/setting/system/View.vue @@ -61,6 +61,12 @@ isPassword: () => false, data: filterValue('ingress.mode'), }, + { + key: 5, + title: t('setting.system.systemSettingItems.lineageSetting.name'), + isPassword: (item: SystemSetting) => item.settingKey === 'lineage.gravitino.token', + data: filterValue('lineage.'), + }, ]; }); const collapseActive = ref(['1', '2', '3', '4', '5']); diff --git a/streampark-console/streampark-console-webapp/src/views/spark/app/create.vue b/streampark-console/streampark-console-webapp/src/views/spark/app/create.vue index 15177f6cc9..cb6c0b2b93 100644 --- a/streampark-console/streampark-console-webapp/src/views/spark/app/create.vue +++ b/streampark-console/streampark-console-webapp/src/views/spark/app/create.vue @@ -73,6 +73,7 @@ appProperties: values.appProperties, appArgs: values.args, hadoopUser: values.hadoopUser, + lineageEnable: values.lineageEnable ?? false, description: values.description, }; await handleCreateAction(params); @@ -95,6 +96,7 @@ appProperties: values.appProperties, appArgs: values.args, hadoopUser: values.hadoopUser, + lineageEnable: values.lineageEnable ?? false, description: values.description, }); } diff --git a/streampark-console/streampark-console-webapp/src/views/spark/app/edit.vue b/streampark-console/streampark-console-webapp/src/views/spark/app/edit.vue index f0dfcc7572..7b45cf1767 100644 --- a/streampark-console/streampark-console-webapp/src/views/spark/app/edit.vue +++ b/streampark-console/streampark-console-webapp/src/views/spark/app/edit.vue @@ -85,6 +85,7 @@ appProperties: values.appProperties, appArgs: values.args, hadoopUser: values.hadoopUser, + lineageEnable: values.lineageEnable ?? false, description: values.description, }; await handleUpdateAction(params); @@ -108,6 +109,7 @@ appProperties: values.appProperties, appArgs: values.args, hadoopUser: values.hadoopUser, + lineageEnable: values.lineageEnable ?? false, description: values.description, }); } diff --git a/streampark-console/streampark-console-webapp/src/views/spark/app/hooks/useAppFormSchema.tsx b/streampark-console/streampark-console-webapp/src/views/spark/app/hooks/useAppFormSchema.tsx index 1d42995881..4e124a449d 100644 --- a/streampark-console/streampark-console-webapp/src/views/spark/app/hooks/useAppFormSchema.tsx +++ b/streampark-console/streampark-console-webapp/src/views/spark/app/hooks/useAppFormSchema.tsx @@ -14,7 +14,7 @@ * See the License for the specific language governing permissions and * limitations under the License. */ -import { computed, onMounted, ref, unref, type Ref } from 'vue'; +import { computed, h, onMounted, ref, unref, type Ref } from 'vue'; import type { FormSchema } from '/@/components/Form'; import { useI18n } from '/@/hooks/web/useI18n'; import { AppExistsStateEnum, JobTypeEnum, DeployMode } from '/@/enums/sparkEnum'; @@ -238,6 +238,14 @@ export function useSparkSchema(sparkEnvs: Ref) { values?.deployMode == DeployMode.YARN_CLIENT || values?.deployMode == DeployMode.YARN_CLUSTER, }, + { + field: 'lineageEnable', + label: t('spark.app.lineageEnable'), + component: 'Switch', + defaultValue: false, + componentProps: { checkedChildren: 'ON', unCheckedChildren: 'OFF' }, + afterItem: () => h('span', { class: 'pop-tip' }, t('spark.app.lineageEnableTip')), + }, { field: 'yarnQueue', label: t('spark.app.yarnQueue'), diff --git a/streampark-flink/streampark-flink-shims/streampark-flink-shims-base-v2/src/main/java/org/apache/streampark/flink/core/FlinkSqlLineageExtractor.java b/streampark-flink/streampark-flink-shims/streampark-flink-shims-base-v2/src/main/java/org/apache/streampark/flink/core/FlinkSqlLineageExtractor.java new file mode 100644 index 0000000000..7f4196601f --- /dev/null +++ b/streampark-flink/streampark-flink-shims/streampark-flink-shims-base-v2/src/main/java/org/apache/streampark/flink/core/FlinkSqlLineageExtractor.java @@ -0,0 +1,152 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one or more + * contributor license agreements. See the NOTICE file distributed with + * this work for additional information regarding copyright ownership. + * The ASF licenses this file to You under the Apache License, Version 2.0 + * (the "License"); you may not use this file except in compliance with + * the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +package org.apache.streampark.flink.core; + +import org.apache.streampark.common.util.StreamParkLoggerFactory; +import org.apache.streampark.flink.core.lineage.CompiledPlanLineageParser; +import org.apache.streampark.flink.core.lineage.LineagePipeline; +import org.apache.streampark.flink.core.lineage.SqlWithOptionsParser; + +import org.apache.streampark.shaded.org.slf4j.Logger; + +import org.apache.flink.configuration.ExecutionOptions; +import org.apache.flink.table.api.CompiledPlan; +import org.apache.flink.table.api.EnvironmentSettings; +import org.apache.flink.table.api.StatementSet; +import org.apache.flink.table.api.TableEnvironment; + +import java.util.ArrayList; +import java.util.LinkedHashMap; +import java.util.List; +import java.util.Map; + +/** + * Flink 2.x counterpart of {@code streampark-flink-shims-base}'s class of the same name — see that + * one's javadoc for the full rationale (classloader/connector-classpath precondition, why this + * never throws, why {@code compilePlan()} is safe to call without ever calling {@code execute()}, + * and why the environment is always a streaming one regardless of the job's declared runtime mode). + * + *

Duplicated rather than shared because {@code SqlCommand} is itself duplicated per-Flink-major + * version in this codebase (see this module's own {@code SqlCommand.java}) — same pattern this + * module's {@code FlinkSqlExecutor} already follows for the same reason: the two variants' SQL + * command regexes are declared independently even where they happen to coincide today. + */ +public final class FlinkSqlLineageExtractor { + + private static final Logger LOG = + StreamParkLoggerFactory.loggerFactory().getLogger(FlinkSqlLineageExtractor.class.getName()); + + private FlinkSqlLineageExtractor() { + } + + public static List extractLineage(String sql) { + try { + return doExtract(sql); + } catch (Exception e) { + LOG.warn("[lineage] failed to extract lineage, submission proceeds without lineage for it", e); + return new ArrayList<>(); + } + } + + /** + * Visible for testing: {@link #extractLineage} swallows every failure by design, so only this + * method can tell "the plan compiled and yielded nothing resolvable" apart from "the plan failed + * to compile at all". + */ + static List doExtract(String sql) throws Exception { + List calls = SqlCommandParser.parseSQL(sql, null); + if (calls == null || calls.isEmpty()) { + return new ArrayList<>(); + } + + TableEnvironment tableEnv = + TableEnvironment.create(EnvironmentSettings.newInstance().inStreamingMode().build()); + StatementSet statementSet = tableEnv.createStatementSet(); + Map tempTables = new LinkedHashMap<>(); + Map catalogTypes = new LinkedHashMap<>(); + boolean hasInsert = false; + + for (SqlCommandCall call : calls) { + switch (call.command) { + case INSERT: + statementSet.addInsertSql(call.originSql); + hasInsert = true; + break; + case SET: + applySet(call, tableEnv); + break; + case SELECT: + case SHOW_CATALOGS: + case SHOW_CURRENT_CATALOG: + case SHOW_DATABASES: + case SHOW_CURRENT_DATABASE: + case SHOW_TABLES: + case SHOW_CREATE_TABLE: + case SHOW_COLUMNS: + case SHOW_VIEWS: + case SHOW_CREATE_VIEW: + case SHOW_FUNCTIONS: + case SHOW_MODULES: + case DESC: + case DESCRIBE: + case EXPLAIN: + case DELETE: + case UPDATE: + case RESET: + case RESET_ALL: + case BEGIN_STATEMENT_SET: + case END_STATEMENT_SET: + // Irrelevant to schema registration or lineage; skip rather than risk a + // side effect (e.g. EXPLAIN executing for real) in a throwaway environment. + break; + default: + if (call.command == SqlCommand.CREATE_TABLE) { + SqlWithOptionsParser.WithOptions options = SqlWithOptionsParser.parse(call.originSql); + if (options != null) { + tempTables.put(options.name(), options); + } + } else if (call.command == SqlCommand.CREATE_CATALOG) { + SqlWithOptionsParser.rememberCatalogType(call.originSql, catalogTypes); + } + tableEnv.executeSql(call.originSql); + } + } + + if (!hasInsert) { + return new ArrayList<>(); + } + + CompiledPlan plan = statementSet.compilePlan(); + return new ArrayList<>(CompiledPlanLineageParser.parse(plan.asJsonString(), tempTables, catalogTypes)); + } + + /** + * Drops the job's {@code execution.runtime-mode}: it can only be chosen when the {@link + * TableEnvironment} is instantiated (Flink rejects any later change, even to the same value), + * and {@link #doExtract} always instantiates a streaming one. + */ + private static void applySet(SqlCommandCall call, TableEnvironment tableEnv) { + if (call.operands == null || call.operands.length < 2) { + return; + } + if (ExecutionOptions.RUNTIME_MODE.key().equals(call.operands[0])) { + return; + } + tableEnv.getConfig().getConfiguration().setString(call.operands[0], call.operands[1]); + } +} diff --git a/streampark-flink/streampark-flink-shims/streampark-flink-shims-base-v2/src/test/java/org/apache/streampark/flink/core/FlinkSqlLineageExtractorTest.java b/streampark-flink/streampark-flink-shims/streampark-flink-shims-base-v2/src/test/java/org/apache/streampark/flink/core/FlinkSqlLineageExtractorTest.java new file mode 100644 index 0000000000..cd34294b38 --- /dev/null +++ b/streampark-flink/streampark-flink-shims/streampark-flink-shims-base-v2/src/test/java/org/apache/streampark/flink/core/FlinkSqlLineageExtractorTest.java @@ -0,0 +1,79 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one or more + * contributor license agreements. See the NOTICE file distributed with + * this work for additional information regarding copyright ownership. + * The ASF licenses this file to You under the Apache License, Version 2.0 + * (the "License"); you may not use this file except in compliance with + * the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +package org.apache.streampark.flink.core; + +import org.junit.jupiter.api.Test; + +import static org.assertj.core.api.Assertions.assertThat; +import static org.assertj.core.api.Assertions.assertThatCode; + +/** + * The Flink 2.x counterpart of the shims-base test of the same name. The two extractors are + * separate classes because the {@code TableEnvironment} bootstrap differs between Flink 1.x and + * 2.x, so each needs its own coverage — a regression fixed in one does not protect the other. + */ +class FlinkSqlLineageExtractorTest { + + /** + * A job declaring {@code execution.runtime-mode = BATCH} must still get a compiled plan. Flink + * implements {@code compilePlan()} only in its stream planner — the batch one throws + * {@code UnsupportedOperationException("The compiled plan feature is not supported in batch + * mode.")} — so honouring that SET when building the throwaway extraction environment loses + * lineage for every batch job. + * + *

Asserted through {@code doExtract} rather than {@link + * FlinkSqlLineageExtractor#extractLineage}: the latter is fail-open and would return the same + * empty list whether the plan compiled or blew up, hiding exactly this regression. + */ + @Test + void batchRuntimeModeDeclarationStillCompilesAPlan() { + assertThatCode(() -> FlinkSqlLineageExtractor.doExtract(batchDeclaringSql())) + .doesNotThrowAnyException(); + } + + /** + * Same SQL through the public entry point: the connectors it uses have no dataset-identity rule, + * so the outcome is no lineage — reported as an empty list, never an exception, because this + * runs on the job submission path. + */ + @Test + void unknownConnectorsYieldNoLineageRatherThanFailing() { + assertThat(FlinkSqlLineageExtractor.extractLineage(batchDeclaringSql())).isEmpty(); + } + + @Test + void sqlWithoutInsertYieldsNoLineage() { + String sql = + "CREATE TABLE probe_src (id BIGINT) WITH (" + + "'connector' = 'datagen'," + + "'number-of-rows' = '1')"; + + assertThat(FlinkSqlLineageExtractor.extractLineage(sql)).isEmpty(); + } + + /** datagen -> blackhole: the only source/sink pair guaranteed present in a bare Flink install. */ + private static String batchDeclaringSql() { + return "SET 'execution.runtime-mode' = 'BATCH';\n" + + "CREATE TABLE probe_src (id BIGINT) WITH (" + + "'connector' = 'datagen'," + + "'number-of-rows' = '1');\n" + + "CREATE TABLE probe_sink (id BIGINT) WITH (" + + "'connector' = 'blackhole');\n" + + "INSERT INTO probe_sink SELECT id FROM probe_src;"; + } +} diff --git a/streampark-flink/streampark-flink-shims/streampark-flink-shims-base/pom.xml b/streampark-flink/streampark-flink-shims/streampark-flink-shims-base/pom.xml index c7734c6d8e..ed03846a5a 100644 --- a/streampark-flink/streampark-flink-shims/streampark-flink-shims-base/pom.xml +++ b/streampark-flink/streampark-flink-shims/streampark-flink-shims-base/pom.xml @@ -81,6 +81,12 @@ hadoop-client-runtime true + + + + com.fasterxml.jackson.core + jackson-databind + diff --git a/streampark-flink/streampark-flink-shims/streampark-flink-shims-base/src/main/java/org/apache/streampark/flink/core/FlinkSqlLineageExtractor.java b/streampark-flink/streampark-flink-shims/streampark-flink-shims-base/src/main/java/org/apache/streampark/flink/core/FlinkSqlLineageExtractor.java new file mode 100644 index 0000000000..b0b32e1dd0 --- /dev/null +++ b/streampark-flink/streampark-flink-shims/streampark-flink-shims-base/src/main/java/org/apache/streampark/flink/core/FlinkSqlLineageExtractor.java @@ -0,0 +1,173 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one or more + * contributor license agreements. See the NOTICE file distributed with + * this work for additional information regarding copyright ownership. + * The ASF licenses this file to You under the Apache License, Version 2.0 + * (the "License"); you may not use this file except in compliance with + * the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +package org.apache.streampark.flink.core; + +import org.apache.streampark.common.util.StreamParkLoggerFactory; +import org.apache.streampark.flink.core.lineage.CompiledPlanLineageParser; +import org.apache.streampark.flink.core.lineage.LineagePipeline; +import org.apache.streampark.flink.core.lineage.SqlWithOptionsParser; + +import org.apache.streampark.shaded.org.slf4j.Logger; + +import org.apache.flink.configuration.ExecutionOptions; +import org.apache.flink.table.api.CompiledPlan; +import org.apache.flink.table.api.EnvironmentSettings; +import org.apache.flink.table.api.StatementSet; +import org.apache.flink.table.api.TableEnvironment; + +import java.util.ArrayList; +import java.util.LinkedHashMap; +import java.util.List; +import java.util.Map; + +/** + * Extracts table-level lineage from a Flink SQL job's source text, without submitting it. + * + *

Called from the Console submission path via {@code FlinkShimsProxy.proxy(...)} — the same + * per-Flink-version, {@code ChildFirstClassLoader}-isolated mechanism other shims features use. + * That classloader is built from the registered Flink Home's entire {@code lib/} directory (see + * {@code FlinkShimsProxy.getFlinkShimsClassLoader}), so a connector-backed {@code CREATE TABLE} + * (mysql-cdc, doris, ...) resolves here exactly when its factory jar is present there — the same + * requirement that already applies for that connector to run for real. If the operator's Flink + * Home lib/ has drifted from the cluster's, this degrades to no lineage for that job (see {@link + * #extractLineage}), not a submission failure. + * + *

Builds a throwaway {@link TableEnvironment} to run every non-{@code INSERT} statement (DDL: + * {@code CREATE TABLE/CATALOG/DATABASE/VIEW}, {@code USE}, ...) for schema registration, collects + * {@code INSERT} statements into a {@link StatementSet}, and calls {@code compilePlan()} — a + * planning-only operation that builds the job graph and validates types without starting any task + * or touching a connector's actual I/O (that only happens on {@code execute()}, which this class + * never calls). + * + *

That environment is always a streaming one, whatever {@code execution.runtime-mode} + * the job itself declares: {@code compilePlan()} is implemented only by Flink's stream planner — + * its batch planner throws {@code UnsupportedOperationException("The compiled plan feature is not + * supported in batch mode.")} (verified in both 1.20 and 2.2). Which tables feed which is a + * property of the query, not of the runtime mode, so planning a batch job's statements as + * streaming yields the same source/sink topology. A statement that genuinely cannot be planned as + * streaming degrades to no lineage for that job, like any other extraction failure. + */ +public final class FlinkSqlLineageExtractor { + + private static final Logger LOG = + StreamParkLoggerFactory.loggerFactory().getLogger(FlinkSqlLineageExtractor.class.getName()); + + private FlinkSqlLineageExtractor() { + } + + /** + * Never throws — this runs on the job submission path, where a lineage gap must never fail the + * submission. Returns an empty list on any failure (logged), including: the SQL has no INSERT, + * a DDL statement fails (e.g. a missing connector factory), or the plan fails to compile. + */ + public static List extractLineage(String sql) { + try { + return doExtract(sql); + } catch (Exception e) { + LOG.warn("[lineage] failed to extract lineage, submission proceeds without lineage for it", e); + return new ArrayList<>(); + } + } + + /** + * Visible for testing: {@link #extractLineage} swallows every failure by design, so only this + * method can tell "the plan compiled and yielded nothing resolvable" apart from "the plan failed + * to compile at all". + */ + static List doExtract(String sql) throws Exception { + List calls = SqlCommandParser.parseSQL(sql, null); + if (calls == null || calls.isEmpty()) { + return new ArrayList<>(); + } + + TableEnvironment tableEnv = + TableEnvironment.create(EnvironmentSettings.newInstance().inStreamingMode().build()); + StatementSet statementSet = tableEnv.createStatementSet(); + Map tempTables = new LinkedHashMap<>(); + Map catalogTypes = new LinkedHashMap<>(); + boolean hasInsert = false; + + for (SqlCommandCall call : calls) { + switch (call.command) { + case INSERT: + statementSet.addInsertSql(call.originSql); + hasInsert = true; + break; + case SET: + applySet(call, tableEnv); + break; + case SELECT: + case SHOW_CATALOGS: + case SHOW_CURRENT_CATALOG: + case SHOW_DATABASES: + case SHOW_CURRENT_DATABASE: + case SHOW_TABLES: + case SHOW_CREATE_TABLE: + case SHOW_COLUMNS: + case SHOW_VIEWS: + case SHOW_CREATE_VIEW: + case SHOW_FUNCTIONS: + case SHOW_MODULES: + case DESC: + case DESCRIBE: + case EXPLAIN: + case DELETE: + case UPDATE: + case RESET: + case RESET_ALL: + case BEGIN_STATEMENT_SET: + case END_STATEMENT_SET: + // Irrelevant to schema registration or lineage; skip rather than risk a + // side effect (e.g. EXPLAIN executing for real) in a throwaway environment. + break; + default: + if (call.command == SqlCommand.CREATE_TABLE) { + SqlWithOptionsParser.WithOptions options = SqlWithOptionsParser.parse(call.originSql); + if (options != null) { + tempTables.put(options.name(), options); + } + } else if (call.command == SqlCommand.CREATE_CATALOG) { + SqlWithOptionsParser.rememberCatalogType(call.originSql, catalogTypes); + } + tableEnv.executeSql(call.originSql); + } + } + + if (!hasInsert) { + return new ArrayList<>(); + } + + CompiledPlan plan = statementSet.compilePlan(); + return new ArrayList<>(CompiledPlanLineageParser.parse(plan.asJsonString(), tempTables, catalogTypes)); + } + + /** + * Drops the job's {@code execution.runtime-mode}: it can only be chosen when the {@link + * TableEnvironment} is instantiated (Flink rejects any later change, even to the same value), + * and {@link #doExtract} always instantiates a streaming one. + */ + private static void applySet(SqlCommandCall call, TableEnvironment tableEnv) { + if (call.operands == null || call.operands.length < 2) { + return; + } + if (ExecutionOptions.RUNTIME_MODE.key().equals(call.operands[0])) { + return; + } + tableEnv.getConfig().getConfiguration().setString(call.operands[0], call.operands[1]); + } +} diff --git a/streampark-flink/streampark-flink-shims/streampark-flink-shims-base/src/main/java/org/apache/streampark/flink/core/lineage/CompiledPlanLineageParser.java b/streampark-flink/streampark-flink-shims/streampark-flink-shims-base/src/main/java/org/apache/streampark/flink/core/lineage/CompiledPlanLineageParser.java new file mode 100644 index 0000000000..029857837a --- /dev/null +++ b/streampark-flink/streampark-flink-shims/streampark-flink-shims-base/src/main/java/org/apache/streampark/flink/core/lineage/CompiledPlanLineageParser.java @@ -0,0 +1,251 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one or more + * contributor license agreements. See the NOTICE file distributed with + * this work for additional information regarding copyright ownership. + * The ASF licenses this file to You under the Apache License, Version 2.0 + * (the "License"); you may not use this file except in compliance with + * the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +package org.apache.streampark.flink.core.lineage; + +import org.apache.streampark.common.util.StreamParkLoggerFactory; + +import org.apache.streampark.shaded.org.slf4j.Logger; + +import com.fasterxml.jackson.databind.JsonNode; +import com.fasterxml.jackson.databind.ObjectMapper; + +import java.util.ArrayDeque; +import java.util.ArrayList; +import java.util.Deque; +import java.util.LinkedHashMap; +import java.util.LinkedHashSet; +import java.util.List; +import java.util.Map; +import java.util.Set; +import java.util.regex.Matcher; +import java.util.regex.Pattern; + +/** + * Resolves the physical OpenLineage dataset identity for every source/sink table in a Flink {@code + * CompiledPlan}, paired per sink rather than flattened across the whole plan (see {@link + * LineagePipeline} for why per-sink pairing matters for {@code STATEMENT SET} jobs). + * + *

Pure Jackson, no Flink API — the plan is already a JSON string by the time it reaches this + * class, produced by {@code StatementSet.compilePlan().asJsonString()} in the per-Flink-version + * extractor. + */ +public final class CompiledPlanLineageParser { + + private static final Logger LOG = + StreamParkLoggerFactory.loggerFactory().getLogger(CompiledPlanLineageParser.class.getName()); + + private static final ObjectMapper MAPPER = new ObjectMapper(); + private static final Pattern IDENTIFIER = Pattern.compile("`([^`]+)`\\.`([^`]+)`\\.`([^`]+)`"); + + /** Plan JSON field names shared by the source, sink and lookup-source table descriptors. */ + private static final String FIELD_TABLE = "table"; + private static final String FIELD_IDENTIFIER = "identifier"; + + private CompiledPlanLineageParser() { + } + + /** + * @param compiledPlanJson the CompiledPlan JSON string + * @param tempTables tables declared in this job's own SQL text via {@code CREATE [TEMPORARY] + * TABLE ... WITH (...)}, keyed by local (unqualified) table name — see {@link + * SqlWithOptionsParser} + * @param catalogTypes the {@code 'type'} of every catalog this job's own SQL text attached via + * {@code CREATE CATALOG}, keyed by catalog name — see {@link #resolveOne} + * @return one {@link LineagePipeline} per sink found in the plan; a sink or input whose + * identity cannot be resolved is dropped (logged), never thrown — see class javadoc on + * {@link DatasetIdentityRegistry} for the fail-open rationale + * @throws IllegalArgumentException if {@code compiledPlanJson} itself is not valid JSON — that + * is a structural/version-mismatch problem worth surfacing, not a per-dataset gap + */ + public static List parse( + String compiledPlanJson, + Map tempTables, + Map catalogTypes) { + JsonNode root; + try { + root = MAPPER.readTree(compiledPlanJson); + } catch (Exception e) { + throw new IllegalArgumentException("Failed to parse CompiledPlan JSON", e); + } + + Map nodesById = indexNodes(root); + Map> predecessors = indexPredecessors(root); + + List pipelines = new ArrayList<>(); + for (JsonNode node : nodesById.values()) { + LineagePipeline pipeline = resolvePipeline(node, nodesById, predecessors, tempTables, catalogTypes); + if (pipeline != null) { + pipelines.add(pipeline); + } + } + return pipelines; + } + + private static Map indexNodes(JsonNode root) { + Map nodesById = new LinkedHashMap<>(); + for (JsonNode node : root.path("nodes")) { + nodesById.put(node.path("id").asInt(), node); + } + return nodesById; + } + + private static Map> indexPredecessors(JsonNode root) { + Map> predecessors = new LinkedHashMap<>(); + for (JsonNode edge : root.path("edges")) { + int source = edge.path("source").asInt(); + int target = edge.path("target").asInt(); + predecessors.computeIfAbsent(target, k -> new ArrayList<>()).add(source); + } + return predecessors; + } + + /** + * The pipeline this node produces, or {@code null} when the node is not a sink at all or its + * sink identity cannot be resolved — in either case it contributes no lineage. + */ + private static LineagePipeline resolvePipeline( + JsonNode node, + Map nodesById, + Map> predecessors, + Map tempTables, + Map catalogTypes) { + JsonNode sink = node.path("dynamicTableSink").path(FIELD_TABLE).path(FIELD_IDENTIFIER); + if (!sink.isTextual()) { + return null; + } + LineageDataset output = resolveOne(sink.asText(), tempTables, catalogTypes, "output"); + if (output == null) { + return null; + } + Set inputs = + collectInputs(node.path("id").asInt(), nodesById, predecessors, tempTables, catalogTypes); + return new LineagePipeline(output, inputs); + } + + /** Every source table reachable upstream of one sink node, walking the plan's edges backwards. */ + private static Set collectInputs( + int sinkNodeId, + Map nodesById, + Map> predecessors, + Map tempTables, + Map catalogTypes) { + Set inputs = new LinkedHashSet<>(); + Set visited = new LinkedHashSet<>(); + Deque pending = new ArrayDeque<>(); + pending.push(sinkNodeId); + while (!pending.isEmpty()) { + int currentId = pending.pop(); + if (!visited.add(currentId)) { + continue; + } + JsonNode current = nodesById.get(currentId); + if (current == null) { + // An edge referencing a node absent from "nodes" would be a malformed plan; + // skip it rather than NPE on a path contracted to only throw on bad JSON. + LOG.warn("[lineage] CompiledPlan edge references unknown node id {}, skipping it", currentId); + } else { + addInput(current.path("scanTableSource").path(FIELD_TABLE).path(FIELD_IDENTIFIER), + inputs, tempTables, catalogTypes); + // Lookup joins (stream-exec-lookup-join) don't produce a scanTableSource node — the + // temporal table they read is nested under temporalTable.lookupTableSource instead. + addInput( + current.path("temporalTable").path("lookupTableSource").path(FIELD_TABLE).path(FIELD_IDENTIFIER), + inputs, tempTables, catalogTypes); + for (int predecessorId : predecessors.getOrDefault(currentId, List.of())) { + pending.push(predecessorId); + } + } + } + return inputs; + } + + private static void addInput( + JsonNode identifier, + Set inputs, + Map tempTables, + Map catalogTypes) { + if (!identifier.isTextual()) { + return; + } + LineageDataset input = resolveOne(identifier.asText(), tempTables, catalogTypes, "input"); + if (input != null) { + inputs.add(input); + } + } + + /** + * Resolves one {@code `catalog`.`database`.`table`} plan identifier to a dataset identity, by + * whichever of the two routes applies: + * + *

+ * + *

Returns {@code null} (logged) when neither applies — a table from a catalog attached + * outside this job's SQL, whose kind is therefore unknowable here. Guessing a scheme would be + * worse than reporting nothing: dataset identity is deduplicated by exact string match, so a + * wrong guess silently splits one physical table into two nodes in the graph instead of failing + * loudly. Same rationale as {@link DatasetIdentityRegistry}'s refusal of a generic fallback. + */ + private static LineageDataset resolveOne( + String rawIdentifier, + Map tempTables, + Map catalogTypes, + String role) { + Matcher matcher = IDENTIFIER.matcher(rawIdentifier); + if (!matcher.matches()) { + LOG.warn("[lineage] unexpected CompiledPlan table identifier shape ({}): {}", role, rawIdentifier); + return null; + } + String catalog = matcher.group(1); + String database = matcher.group(2); + String table = matcher.group(3); + + SqlWithOptionsParser.WithOptions tableOptions = tempTables.get(table); + if (tableOptions != null) { + LineageDataset resolved = DatasetIdentityRegistry.resolve(table, tableOptions.options()); + if (resolved != null) { + LOG.info( + "[lineage] dataset resolved from WITH options (connector={}, role={}): {}", + tableOptions.options().get("connector"), + role, + resolved); + } + return resolved; + } + + String catalogType = catalogTypes.get(catalog); + if (catalogType == null) { + LOG.warn( + "[lineage] table `{}` belongs to catalog `{}`, whose type is not declared in this job's SQL" + + " (role={}), skipping lineage for it", + table, + catalog, + role); + return null; + } + LineageDataset resolved = new LineageDataset(catalogType + "://" + catalog + "/" + database, table); + LOG.info("[lineage] dataset resolved from catalog identifier (role={}): {}", role, resolved); + return resolved; + } +} diff --git a/streampark-flink/streampark-flink-shims/streampark-flink-shims-base/src/main/java/org/apache/streampark/flink/core/lineage/DatasetIdentityRegistry.java b/streampark-flink/streampark-flink-shims/streampark-flink-shims-base/src/main/java/org/apache/streampark/flink/core/lineage/DatasetIdentityRegistry.java new file mode 100644 index 0000000000..278ab6c7d1 --- /dev/null +++ b/streampark-flink/streampark-flink-shims/streampark-flink-shims-base/src/main/java/org/apache/streampark/flink/core/lineage/DatasetIdentityRegistry.java @@ -0,0 +1,104 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one or more + * contributor license agreements. See the NOTICE file distributed with + * this work for additional information regarding copyright ownership. + * The ASF licenses this file to You under the Apache License, Version 2.0 + * (the "License"); you may not use this file except in compliance with + * the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +package org.apache.streampark.flink.core.lineage; + +import org.apache.streampark.common.util.StreamParkLoggerFactory; + +import org.apache.streampark.shaded.org.slf4j.Logger; + +import java.util.Map; + +/** + * Maps a Flink connector's {@code WITH (...)} options to the OpenLineage dataset identity Gravitino + * expects — the {@code (namespace, name)} pair must be byte-identical to what other emitters into + * the same Gravitino graph produce for the same physical table, so these rules are not invented + * here; they mirror the existing convention already in use for this Gravitino deployment. + * + *

Fail-open by design: an unknown connector or a connector missing a required option logs a + * {@code WARN} and returns {@code null} rather than throwing. This runs on the job submission path + * — a lineage gap must never fail the submission. + * + *

To support another connector, add a case below with its own {@code (namespace, name)} rule; + * do not guess a generic fallback. + */ +public final class DatasetIdentityRegistry { + + private static final Logger LOG = + StreamParkLoggerFactory.loggerFactory().getLogger(DatasetIdentityRegistry.class.getName()); + + private DatasetIdentityRegistry() { + } + + /** + * Resolves the dataset identity for a table declared via {@code CREATE [TEMPORARY] TABLE ... + * WITH (...)}. Returns {@code null} (logging why) when the connector is unrecognized or a + * required option is missing. + */ + public static LineageDataset resolve(String tableName, Map options) { + String connector = options.get("connector"); + if (connector == null) { + LOG.warn( + "[lineage] table `{}` has no 'connector' WITH option, skipping lineage for it", + tableName); + return null; + } + switch (connector) { + case "mysql-cdc": + return resolveMysqlCdc(tableName, options); + case "doris": + return resolveDoris(tableName, options); + default: + LOG.warn( + "[lineage] no dataset-identity rule for connector '{}' on table `{}`, skipping lineage for it", + connector, + tableName); + return null; + } + } + + private static LineageDataset resolveMysqlCdc(String tableName, Map options) { + String hostname = require(tableName, options, "hostname"); + String port = require(tableName, options, "port"); + String database = require(tableName, options, "database-name"); + String table = require(tableName, options, "table-name"); + if (hostname == null || port == null || database == null || table == null) { + return null; + } + return new LineageDataset("mysql-cdc://" + hostname + ":" + port, database + "." + table); + } + + private static LineageDataset resolveDoris(String tableName, Map options) { + String fenodes = require(tableName, options, "fenodes"); + String identifier = require(tableName, options, "table.identifier"); + if (fenodes == null || identifier == null) { + return null; + } + return new LineageDataset("doris://" + fenodes, identifier); + } + + private static String require(String tableName, Map options, String key) { + String value = options.get(key); + if (value == null) { + LOG.warn( + "[lineage] table `{}` is missing required WITH option '{}', skipping lineage for it", + tableName, + key); + } + return value; + } +} diff --git a/streampark-flink/streampark-flink-shims/streampark-flink-shims-base/src/main/java/org/apache/streampark/flink/core/lineage/LineageDataset.java b/streampark-flink/streampark-flink-shims/streampark-flink-shims-base/src/main/java/org/apache/streampark/flink/core/lineage/LineageDataset.java new file mode 100644 index 0000000000..2953807d1f --- /dev/null +++ b/streampark-flink/streampark-flink-shims/streampark-flink-shims-base/src/main/java/org/apache/streampark/flink/core/lineage/LineageDataset.java @@ -0,0 +1,75 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one or more + * contributor license agreements. See the NOTICE file distributed with + * this work for additional information regarding copyright ownership. + * The ASF licenses this file to You under the Apache License, Version 2.0 + * (the "License"); you may not use this file except in compliance with + * the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +package org.apache.streampark.flink.core.lineage; + +import java.io.Serializable; +import java.util.Objects; + +/** + * OpenLineage dataset identity as sent to Gravitino's {@code POST /api/lineage}. + * + *

Equality is exact string match on (namespace, name) — that is what Gravitino's {@code + * lineage_dataset} table dedupes on, so the same physical table must resolve to a byte-identical + * identity everywhere it is produced or consumed, including by other non-StreamPark emitters + * writing into the same Gravitino graph. + * + *

Must remain {@link Serializable}: instances cross Flink-version classloader boundaries via + * {@code FlinkShimsProxy.getObject}, which round-trips objects through Java serialization. + */ +public final class LineageDataset implements Serializable { + + private static final long serialVersionUID = 1L; + + private final String namespace; + private final String name; + + public LineageDataset(String namespace, String name) { + this.namespace = namespace; + this.name = name; + } + + public String namespace() { + return namespace; + } + + public String name() { + return name; + } + + @Override + public boolean equals(Object o) { + if (this == o) { + return true; + } + if (!(o instanceof LineageDataset)) { + return false; + } + LineageDataset that = (LineageDataset) o; + return Objects.equals(namespace, that.namespace) && Objects.equals(name, that.name); + } + + @Override + public int hashCode() { + return Objects.hash(namespace, name); + } + + @Override + public String toString() { + return namespace + "/" + name; + } +} diff --git a/streampark-flink/streampark-flink-shims/streampark-flink-shims-base/src/main/java/org/apache/streampark/flink/core/lineage/LineagePipeline.java b/streampark-flink/streampark-flink-shims/streampark-flink-shims-base/src/main/java/org/apache/streampark/flink/core/lineage/LineagePipeline.java new file mode 100644 index 0000000000..1d0ca9bb08 --- /dev/null +++ b/streampark-flink/streampark-flink-shims/streampark-flink-shims-base/src/main/java/org/apache/streampark/flink/core/lineage/LineagePipeline.java @@ -0,0 +1,58 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one or more + * contributor license agreements. See the NOTICE file distributed with + * this work for additional information regarding copyright ownership. + * The ASF licenses this file to You under the Apache License, Version 2.0 + * (the "License"); you may not use this file except in compliance with + * the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +package org.apache.streampark.flink.core.lineage; + +import java.io.Serializable; +import java.util.Collections; +import java.util.LinkedHashSet; +import java.util.Set; + +/** + * One sink dataset and exactly the input datasets reachable to it through the compiled plan (not + * to any sibling sink in the same job). + * + *

A {@code STATEMENT SET} with N independent {@code INSERT} statements compiles into N disjoint + * connected components in the plan graph. Flattening all of them into one shared input set and one + * shared output set would produce a full N×M cross product of edges when Gravitino stores the + * resulting OpenLineage events, most of which would be false — table A's data may never actually + * reach table B's sink. Keeping one {@link LineagePipeline} per sink, each with only the inputs its + * own subgraph reaches, avoids that. A job with a single {@code INSERT} degenerates to one pipeline + * with N inputs, same as before. + * + *

Must remain {@link Serializable} for the same reason as {@link LineageDataset}. + */ +public final class LineagePipeline implements Serializable { + + private static final long serialVersionUID = 1L; + + private final LineageDataset output; + private final Set inputs; + + public LineagePipeline(LineageDataset output, Set inputs) { + this.output = output; + this.inputs = Collections.unmodifiableSet(new LinkedHashSet<>(inputs)); + } + + public LineageDataset output() { + return output; + } + + public Set inputs() { + return inputs; + } +} diff --git a/streampark-flink/streampark-flink-shims/streampark-flink-shims-base/src/main/java/org/apache/streampark/flink/core/lineage/SqlWithOptionsParser.java b/streampark-flink/streampark-flink-shims/streampark-flink-shims-base/src/main/java/org/apache/streampark/flink/core/lineage/SqlWithOptionsParser.java new file mode 100644 index 0000000000..c6636b8a9a --- /dev/null +++ b/streampark-flink/streampark-flink-shims/streampark-flink-shims-base/src/main/java/org/apache/streampark/flink/core/lineage/SqlWithOptionsParser.java @@ -0,0 +1,242 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one or more + * contributor license agreements. See the NOTICE file distributed with + * this work for additional information regarding copyright ownership. + * The ASF licenses this file to You under the Apache License, Version 2.0 + * (the "License"); you may not use this file except in compliance with + * the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +package org.apache.streampark.flink.core.lineage; + +import java.util.LinkedHashMap; +import java.util.Map; +import java.util.regex.Matcher; +import java.util.regex.Pattern; + +/** + * Extracts the declared name and {@code WITH (...)} options from one {@code CREATE + * [TEMPORARY] TABLE} or {@code CREATE CATALOG} statement. + * + *

Why this exists: a Flink {@code CompiledPlan} gives an accurate {@code + * `catalog`.`database`.`table`} identifier only for tables that live in a real, attached Flink + * catalog (e.g. a Paimon catalog registered via {@code CREATE CATALOG}). For a table created + * per-job via {@code CREATE TABLE ... WITH (...)} — the common shape for StreamPark Flink SQL jobs, + * which are typically self-contained scripts with no attached catalog — the plan JSON reports only + * "session default catalog/database + local table name", carrying no connector/host/physical-table + * info at all. That information lives in the SQL text itself, so it is extracted here instead. + * + *

The same holds one level up for {@code CREATE CATALOG ... WITH ('type' = '...')} — see + * {@link CompiledPlanLineageParser#resolveOne} for why a catalog's declared type has to come from + * the SQL text as well. + * + *

Deliberately not scoped to {@code TEMPORARY} tables only (unlike the reference implementation + * this was ported from): a plain {@code CREATE TABLE} without an attached persistent catalog is + * exactly as ephemeral as a {@code CREATE TEMPORARY TABLE} from Gravitino's point of view, and + * StreamPark SQL jobs commonly omit the {@code TEMPORARY} keyword. Restricting to {@code TEMPORARY} + * would silently lose lineage for the common case. + */ +public final class SqlWithOptionsParser { + + /** + * The alternatives of a single SQL identifier, bare or backtick-quoted (a quoted one may contain + * dots). Deliberately ungrouped, so each use can wrap it in whichever kind of group it needs + * without nesting a redundant one inside. Both letter cases are spelled out, so the pattern + * using it must not be compiled {@code CASE_INSENSITIVE} — only the keyword patterns are, which + * also keeps identifier matching case-exact, as Flink treats it. + */ + private static final String IDENTIFIER = "`[^`]+`|[A-Za-z_][A-Za-z0-9_$]*"; + + /** The {@code CREATE ... TABLE} keywords, up to where the declared name starts. */ + private static final Pattern CREATE_TABLE_KEYWORDS = + Pattern.compile( + "\\s*CREATE\\s+(?:TEMPORARY\\s+)?TABLE\\s+(?:IF\\s+NOT\\s+EXISTS\\s+)?", + Pattern.CASE_INSENSITIVE); + + /** The {@code CREATE CATALOG} keywords, up to where the declared name starts. */ + private static final Pattern CREATE_CATALOG_KEYWORDS = + Pattern.compile("\\s*CREATE\\s+CATALOG\\s+(?:IF\\s+NOT\\s+EXISTS\\s+)?", Pattern.CASE_INSENSITIVE); + + /** + * The declared name, matched from just past those keywords. It may be qualified ({@code CREATE + * TABLE mydb.mytable ...}); the qualifier prefix is matched but not captured, so the capture + * group is the local name alone — that is what a {@code CompiledPlan} identifier reports the + * table under. Kept apart from the keywords rather than inlined into both patterns above: one + * definition of what a declared name looks like, and neither pattern then has to carry the + * other's share of the complexity. + * + *

The qualifier repetition is possessive: every iteration ends in a dot, so no match ever + * needs to backtrack into an already-accepted qualifier, and a possessive group repetition is + * matched iteratively rather than recursively (no stack growth proportional to the identifier + * count). + */ + private static final Pattern DECLARED_NAME = + Pattern.compile("(?:(?:" + IDENTIFIER + ")\\s*\\.\\s*)*+(" + IDENTIFIER + ")"); + + private static final Pattern WITH_CLAUSE = Pattern.compile("\\bWITH\\s*\\(", Pattern.CASE_INSENSITIVE); + /** + * {@code 'key' = 'value'}. A literal's embedded {@code ''} must count as an escaped single quote, + * not the end of the literal — a connector option value (e.g. a password) may legitimately + * contain a quote escaped this way, and treating it as the literal's end would truncate the + * value at that point. + * + *

Both literal bodies repeat possessively. Nothing inside a literal can match the closing + * quote that follows it (a {@code '} is only ever consumed as part of {@code ''} or {@code \'}), + * so a well-formed entry never needs to give characters back; only an unterminated literal now + * fails to match instead of being salvaged into a truncated option, which is the better outcome + * for a lenient parser. The gain is that a possessive group repetition is matched iteratively — + * a greedy one recurses once per character, overflowing the stack on a long option value. + */ + private static final Pattern OPTION_ENTRY = + Pattern.compile("'((?:[^'\\\\]|\\\\.|'')*+)'\\s*=\\s*'((?:[^'\\\\]|\\\\.|'')*+)'"); + + private SqlWithOptionsParser() { + } + + /** One declaration's local (unqualified) name and its {@code WITH (...)} options. */ + public static final class WithOptions { + + private final String name; + private final Map options; + + WithOptions(String name, Map options) { + this.name = name; + this.options = options; + } + + public String name() { + return name; + } + + public Map options() { + return options; + } + } + + /** + * Parses one {@code CREATE [TEMPORARY] TABLE} statement. Returns {@code null} (not an + * exception — this is production submission-path code, not a validator) when the statement + * does not match the expected shape or carries no {@code WITH (...)} clause, e.g. {@code CREATE + * TABLE ... LIKE ...} or a catalog-backed table with no inline connector options. + */ + public static WithOptions parse(String createTableStatement) { + return parseDeclaration(CREATE_TABLE_KEYWORDS, createTableStatement); + } + + /** + * Parses one {@code CREATE CATALOG} statement, same contract as {@link #parse}. The interesting + * option is {@code 'type'}, which names the catalog implementation (paimon, hive, jdbc, ...). + */ + public static WithOptions parseCatalog(String createCatalogStatement) { + return parseDeclaration(CREATE_CATALOG_KEYWORDS, createCatalogStatement); + } + + /** + * Records a {@code CREATE CATALOG}'s declared {@code 'type'} under its catalog name, or does + * nothing when the statement declares none. Lives here rather than in each per-Flink-version + * extractor that calls it: it is pure SQL-text parsing with no version-specific type in its + * signature, so duplicating it alongside those extractors would only risk them diverging. + */ + public static void rememberCatalogType(String createCatalogStatement, Map catalogTypes) { + WithOptions catalog = parseCatalog(createCatalogStatement); + if (catalog == null) { + return; + } + String type = catalog.options().get("type"); + if (type != null) { + catalogTypes.put(catalog.name(), type); + } + } + + private static WithOptions parseDeclaration(Pattern keywords, String statement) { + Matcher keywordMatcher = keywords.matcher(statement); + if (!keywordMatcher.lookingAt()) { + return null; + } + // Anchored at the keywords' end rather than searched for: a name found anywhere else in the + // statement (a column, a WITH value) would not be the declared one. + Matcher nameMatcher = DECLARED_NAME.matcher(statement).region(keywordMatcher.end(), statement.length()); + if (!nameMatcher.lookingAt()) { + return null; + } + String name = unquote(nameMatcher.group(1)); + + int bodyStart = findWithClauseBodyStart(statement); + if (bodyStart < 0) { + return null; + } + String body = extractParenthesizedBody(statement, bodyStart); + if (body == null) { + return null; + } + + Map options = new LinkedHashMap<>(); + Matcher entry = OPTION_ENTRY.matcher(body); + while (entry.find()) { + options.put(unescapeLiteral(entry.group(1)), unescapeLiteral(entry.group(2))); + } + return new WithOptions(name, options); + } + + /** Strips the backticks around a quoted identifier: {@code `my.table`} to {@code my.table}. */ + private static String unquote(String identifier) { + return identifier.startsWith("`") ? identifier.substring(1, identifier.length() - 1) : identifier; + } + + /** + * Index just past the opening {@code (} of the first {@code WITH (} that is not itself inside a + * string literal, or {@code -1}. The quote check matters: a column {@code COMMENT 'see WITH + * (x)'} would otherwise be mistaken for the options clause and parsed as garbage. + */ + private static int findWithClauseBodyStart(String sql) { + Matcher withStart = WITH_CLAUSE.matcher(sql); + int scanned = 0; + boolean inSingleQuote = false; + while (withStart.find()) { + while (scanned < withStart.start()) { + if (sql.charAt(scanned) == '\'') { + inSingleQuote = !inSingleQuote; + } + scanned++; + } + if (!inSingleQuote) { + return withStart.end(); + } + } + return -1; + } + + /** Balanced-parenthesis scan from just past the opening {@code (}, quote-aware. */ + private static String extractParenthesizedBody(String sql, int bodyStart) { + int depth = 1; + int idx = bodyStart; + boolean inSingleQuote = false; + while (idx < sql.length() && depth > 0) { + char c = sql.charAt(idx); + if (c == '\'') { + inSingleQuote = !inSingleQuote; + } else if (!inSingleQuote && c == '(') { + depth++; + } else if (!inSingleQuote && c == ')') { + depth--; + } + idx++; + } + if (depth != 0) { + return null; + } + return sql.substring(bodyStart, idx - 1); + } + + private static String unescapeLiteral(String literal) { + return literal.replace("''", "'"); + } +} diff --git a/streampark-flink/streampark-flink-shims/streampark-flink-shims-base/src/test/java/org/apache/streampark/flink/core/FlinkSqlLineageExtractorTest.java b/streampark-flink/streampark-flink-shims/streampark-flink-shims-base/src/test/java/org/apache/streampark/flink/core/FlinkSqlLineageExtractorTest.java new file mode 100644 index 0000000000..7ffb0156c9 --- /dev/null +++ b/streampark-flink/streampark-flink-shims/streampark-flink-shims-base/src/test/java/org/apache/streampark/flink/core/FlinkSqlLineageExtractorTest.java @@ -0,0 +1,74 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one or more + * contributor license agreements. See the NOTICE file distributed with + * this work for additional information regarding copyright ownership. + * The ASF licenses this file to You under the Apache License, Version 2.0 + * (the "License"); you may not use this file except in compliance with + * the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +package org.apache.streampark.flink.core; + +import org.junit.jupiter.api.Test; + +import static org.assertj.core.api.Assertions.assertThat; +import static org.assertj.core.api.Assertions.assertThatCode; + +class FlinkSqlLineageExtractorTest { + + /** + * A job declaring {@code execution.runtime-mode = BATCH} must still get a compiled plan. Flink + * implements {@code compilePlan()} only in its stream planner — the batch one throws + * {@code UnsupportedOperationException("The compiled plan feature is not supported in batch + * mode.")} — so honouring that SET when building the throwaway extraction environment loses + * lineage for every batch job. + * + *

Asserted through {@code doExtract} rather than {@link + * FlinkSqlLineageExtractor#extractLineage}: the latter is fail-open and would return the same + * empty list whether the plan compiled or blew up, hiding exactly this regression. + */ + @Test + void batchRuntimeModeDeclarationStillCompilesAPlan() { + assertThatCode(() -> FlinkSqlLineageExtractor.doExtract(batchDeclaringSql())) + .doesNotThrowAnyException(); + } + + /** + * Same SQL through the public entry point: the connectors it uses have no dataset-identity rule, + * so the outcome is no lineage — reported as an empty list, never an exception, because this + * runs on the job submission path. + */ + @Test + void unknownConnectorsYieldNoLineageRatherThanFailing() { + assertThat(FlinkSqlLineageExtractor.extractLineage(batchDeclaringSql())).isEmpty(); + } + + @Test + void sqlWithoutInsertYieldsNoLineage() { + String sql = + "CREATE TABLE probe_src (id BIGINT) WITH (" + + "'connector' = 'datagen'," + + "'number-of-rows' = '1')"; + + assertThat(FlinkSqlLineageExtractor.extractLineage(sql)).isEmpty(); + } + + /** datagen -> blackhole: the only source/sink pair guaranteed present in a bare Flink install. */ + private static String batchDeclaringSql() { + return "SET 'execution.runtime-mode' = 'BATCH';\n" + + "CREATE TABLE probe_src (id BIGINT) WITH (" + + "'connector' = 'datagen'," + + "'number-of-rows' = '1');\n" + + "CREATE TABLE probe_sink (id BIGINT) WITH (" + + "'connector' = 'blackhole');\n" + + "INSERT INTO probe_sink SELECT id FROM probe_src;"; + } +} diff --git a/streampark-flink/streampark-flink-shims/streampark-flink-shims-base/src/test/java/org/apache/streampark/flink/core/lineage/CompiledPlanLineageParserTest.java b/streampark-flink/streampark-flink-shims/streampark-flink-shims-base/src/test/java/org/apache/streampark/flink/core/lineage/CompiledPlanLineageParserTest.java new file mode 100644 index 0000000000..c482ac0c7c --- /dev/null +++ b/streampark-flink/streampark-flink-shims/streampark-flink-shims-base/src/test/java/org/apache/streampark/flink/core/lineage/CompiledPlanLineageParserTest.java @@ -0,0 +1,215 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one or more + * contributor license agreements. See the NOTICE file distributed with + * this work for additional information regarding copyright ownership. + * The ASF licenses this file to You under the Apache License, Version 2.0 + * (the "License"); you may not use this file except in compliance with + * the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +package org.apache.streampark.flink.core.lineage; + +import org.junit.jupiter.api.Test; + +import java.util.List; +import java.util.Map; + +import static org.assertj.core.api.Assertions.assertThat; +import static org.assertj.core.api.Assertions.assertThatThrownBy; + +/** + * Plan JSON shapes below mirror real Flink {@code CompiledPlan.asJsonString()} output (confirmed + * against a real Flink SQL job's plan in the reference implementation this parser was ported from + * — see {@code datasophon-lineage-emitter}'s {@code DatasetResolverTest}), trimmed to only the + * fields this parser reads. + */ +class CompiledPlanLineageParserTest { + + /** + * What {@code CREATE CATALOG paimon_s3 WITH ('type' = 'paimon', ...)} in the job's own SQL + * contributes: the plan identifies a table's catalog by name only, so the catalog's declared + * type is the only thing that says how its datasets should be named. + */ + private static final Map PAIMON_CATALOG = Map.of("paimon_s3", "paimon"); + + private static SqlWithOptionsParser.WithOptions tempTable(String name, Map options) { + SqlWithOptionsParser.WithOptions parsed = + SqlWithOptionsParser.parse( + "CREATE TABLE `" + name + "` WITH (" + toWithClause(options) + ")"); + assertThat(parsed).isNotNull(); + return parsed; + } + + private static String toWithClause(Map options) { + StringBuilder sb = new StringBuilder(); + options.forEach((k, v) -> sb.append("'").append(k).append("'='").append(v).append("',")); + sb.setLength(sb.length() - 1); + return sb.toString(); + } + + @Test + void resolvesMysqlCdcSourceAndPaimonCatalogSink() { + String plan = + "{\"nodes\":[" + + "{\"id\":1,\"scanTableSource\":{\"table\":{\"identifier\":\"`paimon_s3`.`lineage_flink_verify`.`mysql_pat_surgery`\"}}}," + + "{\"id\":2,\"dynamicTableSink\":{\"table\":{\"identifier\":\"`paimon_s3`.`lineage_flink_verify`.`ods_pat_surgery`\"}}}" + + "],\"edges\":[{\"source\":1,\"target\":2}]}"; + SqlWithOptionsParser.WithOptions source = + tempTable( + "mysql_pat_surgery", + Map.of( + "connector", "mysql-cdc", + "hostname", "192.168.10.131", + "port", "3306", + "database-name", "lineage_flink_verify", + "table-name", "pat_surgery")); + + List pipelines = + CompiledPlanLineageParser.parse(plan, Map.of(source.name(), source), PAIMON_CATALOG); + + assertThat(pipelines).hasSize(1); + LineagePipeline pipeline = pipelines.get(0); + assertThat(pipeline.output()) + .isEqualTo(new LineageDataset("paimon://paimon_s3/lineage_flink_verify", "ods_pat_surgery")); + assertThat(pipeline.inputs()) + .containsExactly( + new LineageDataset("mysql-cdc://192.168.10.131:3306", "lineage_flink_verify.pat_surgery")); + } + + @Test + void resolvesLookupJoinTemporalTableAsInput() { + // stream-exec-lookup-join nodes don't have a scanTableSource — the temporal table they + // read is nested under temporalTable.lookupTableSource instead. + String plan = + "{\"nodes\":[" + + "{\"id\":1,\"scanTableSource\":{\"table\":{\"identifier\":\"`paimon_s3`.`db`.`fact`\"}}}," + + "{\"id\":2,\"temporalTable\":{\"lookupTableSource\":{\"table\":{\"identifier\":\"`paimon_s3`.`db`.`dim`\"}}}}," + + "{\"id\":3,\"dynamicTableSink\":{\"table\":{\"identifier\":\"`paimon_s3`.`db`.`sink`\"}}}" + + "],\"edges\":[{\"source\":1,\"target\":2},{\"source\":2,\"target\":3}]}"; + + List pipelines = CompiledPlanLineageParser.parse(plan, Map.of(), PAIMON_CATALOG); + + assertThat(pipelines).hasSize(1); + assertThat(pipelines.get(0).inputs()) + .containsExactlyInAnyOrder( + new LineageDataset("paimon://paimon_s3/db", "fact"), + new LineageDataset("paimon://paimon_s3/db", "dim")); + } + + @Test + void pairsEachSinkWithOnlyItsOwnInputsInAStatementSet() { + // T15-style finding (see LineagePipeline javadoc): a STATEMENT SET compiles into disjoint + // connected components, one per INSERT. Flattening across the whole plan would report + // every input as feeding every output — this plan mirrors that shape with two independent + // two-node pipelines that must not cross-contaminate each other's inputs. + String plan = + "{\"nodes\":[" + + "{\"id\":1,\"scanTableSource\":{\"table\":{\"identifier\":\"`paimon_s3`.`db`.`src_a`\"}}}," + + "{\"id\":2,\"dynamicTableSink\":{\"table\":{\"identifier\":\"`paimon_s3`.`db`.`sink_a`\"}}}," + + "{\"id\":3,\"scanTableSource\":{\"table\":{\"identifier\":\"`paimon_s3`.`db`.`src_b`\"}}}," + + "{\"id\":4,\"dynamicTableSink\":{\"table\":{\"identifier\":\"`paimon_s3`.`db`.`sink_b`\"}}}" + + "],\"edges\":[{\"source\":1,\"target\":2},{\"source\":3,\"target\":4}]}"; + + List pipelines = CompiledPlanLineageParser.parse(plan, Map.of(), PAIMON_CATALOG); + + assertThat(pipelines).hasSize(2); + for (LineagePipeline pipeline : pipelines) { + assertThat(pipeline.inputs()).hasSize(1); + String inputName = pipeline.inputs().iterator().next().name(); + if (pipeline.output().name().equals("sink_a")) { + assertThat(inputName).isEqualTo("src_a"); + } else { + assertThat(inputName).isEqualTo("src_b"); + } + } + } + + @Test + void dropsUnresolvableInputInsteadOfFailingWholePipeline() { + // The sink resolves fine (Paimon catalog table); the source is a per-job table declared + // with an unrecognized connector. Fail-open: the pipeline is still returned, just missing + // that one input — never an exception on the submission path. + String plan = + "{\"nodes\":[" + + "{\"id\":1,\"scanTableSource\":{\"table\":{\"identifier\":\"`paimon_s3`.`db`.`kafka_src`\"}}}," + + "{\"id\":2,\"dynamicTableSink\":{\"table\":{\"identifier\":\"`paimon_s3`.`db`.`sink`\"}}}" + + "],\"edges\":[{\"source\":1,\"target\":2}]}"; + SqlWithOptionsParser.WithOptions source = tempTable("kafka_src", Map.of("connector", "kafka")); + + List pipelines = + CompiledPlanLineageParser.parse(plan, Map.of(source.name(), source), PAIMON_CATALOG); + + assertThat(pipelines).hasSize(1); + assertThat(pipelines.get(0).inputs()).isEmpty(); + } + + @Test + void namespaceSchemeFollowsTheCatalogsDeclaredType() { + // Nothing here is Paimon-specific: the scheme is whatever `CREATE CATALOG ... WITH + // ('type' = ...)` declared, so a Hive catalog's datasets must be named hive://, not + // silently reported under some other engine's namespace. + String plan = + "{\"nodes\":[" + + "{\"id\":1,\"scanTableSource\":{\"table\":{\"identifier\":\"`hive_prod`.`db`.`src`\"}}}," + + "{\"id\":2,\"dynamicTableSink\":{\"table\":{\"identifier\":\"`hive_prod`.`db`.`sink`\"}}}" + + "],\"edges\":[{\"source\":1,\"target\":2}]}"; + + List pipelines = + CompiledPlanLineageParser.parse(plan, Map.of(), Map.of("hive_prod", "hive")); + + assertThat(pipelines).hasSize(1); + assertThat(pipelines.get(0).output()).isEqualTo(new LineageDataset("hive://hive_prod/db", "sink")); + assertThat(pipelines.get(0).inputs()) + .containsExactly(new LineageDataset("hive://hive_prod/db", "src")); + } + + @Test + void skipsCatalogTableWhoseCatalogTypeIsUnknown() { + // A catalog attached outside this job's SQL leaves its type unknowable here. Guessing a + // scheme would silently split one physical table into two graph nodes (dataset identity is + // deduplicated by exact string), which is worse than reporting nothing. + String plan = + "{\"nodes\":[" + + "{\"id\":1,\"scanTableSource\":{\"table\":{\"identifier\":\"`unknown_cat`.`db`.`src`\"}}}," + + "{\"id\":2,\"dynamicTableSink\":{\"table\":{\"identifier\":\"`unknown_cat`.`db`.`sink`\"}}}" + + "],\"edges\":[{\"source\":1,\"target\":2}]}"; + + assertThat(CompiledPlanLineageParser.parse(plan, Map.of(), Map.of())).isEmpty(); + } + + @Test + void ignoresEdgesReferencingNodesAbsentFromThePlan() { + String plan = + "{\"nodes\":[" + + "{\"id\":2,\"dynamicTableSink\":{\"table\":{\"identifier\":\"`paimon_s3`.`db`.`sink`\"}}}" + + "],\"edges\":[{\"source\":99,\"target\":2}]}"; + + List pipelines = CompiledPlanLineageParser.parse(plan, Map.of(), PAIMON_CATALOG); + + assertThat(pipelines).hasSize(1); + assertThat(pipelines.get(0).inputs()).isEmpty(); + } + + @Test + void throwsOnMalformedJsonRatherThanSilentlyReturningEmpty() { + // Malformed CompiledPlan JSON is a structural/version-mismatch problem worth surfacing to + // the caller's own try/catch, not a per-dataset resolution gap to silently absorb here. + assertThatThrownBy(() -> CompiledPlanLineageParser.parse("not json", Map.of(), Map.of())) + .isInstanceOf(IllegalArgumentException.class); + } + + @Test + void plansWithNoSinksYieldNoPipelines() { + String plan = "{\"nodes\":[],\"edges\":[]}"; + + assertThat(CompiledPlanLineageParser.parse(plan, Map.of(), PAIMON_CATALOG)).isEmpty(); + } +} diff --git a/streampark-flink/streampark-flink-shims/streampark-flink-shims-base/src/test/java/org/apache/streampark/flink/core/lineage/DatasetIdentityRegistryTest.java b/streampark-flink/streampark-flink-shims/streampark-flink-shims-base/src/test/java/org/apache/streampark/flink/core/lineage/DatasetIdentityRegistryTest.java new file mode 100644 index 0000000000..279377befa --- /dev/null +++ b/streampark-flink/streampark-flink-shims/streampark-flink-shims-base/src/test/java/org/apache/streampark/flink/core/lineage/DatasetIdentityRegistryTest.java @@ -0,0 +1,75 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one or more + * contributor license agreements. See the NOTICE file distributed with + * this work for additional information regarding copyright ownership. + * The ASF licenses this file to You under the Apache License, Version 2.0 + * (the "License"); you may not use this file except in compliance with + * the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +package org.apache.streampark.flink.core.lineage; + +import org.junit.jupiter.api.Test; + +import java.util.Map; + +import static org.assertj.core.api.Assertions.assertThat; + +class DatasetIdentityRegistryTest { + + @Test + void resolvesMysqlCdc() { + LineageDataset dataset = + DatasetIdentityRegistry.resolve( + "src", + Map.of( + "connector", "mysql-cdc", + "hostname", "192.168.10.131", + "port", "3306", + "database-name", "lineage_flink_verify", + "table-name", "pat_surgery")); + + assertThat(dataset).isNotNull(); + assertThat(dataset.namespace()).isEqualTo("mysql-cdc://192.168.10.131:3306"); + assertThat(dataset.name()).isEqualTo("lineage_flink_verify.pat_surgery"); + } + + @Test + void resolvesDoris() { + LineageDataset dataset = + DatasetIdentityRegistry.resolve( + "sink", + Map.of( + "connector", "doris", + "fenodes", "192.168.10.131:8030", + "table.identifier", "db.ods_table")); + + assertThat(dataset).isNotNull(); + assertThat(dataset.namespace()).isEqualTo("doris://192.168.10.131:8030"); + assertThat(dataset.name()).isEqualTo("db.ods_table"); + } + + @Test + void returnsNullForUnknownConnectorInsteadOfThrowing() { + assertThat(DatasetIdentityRegistry.resolve("t", Map.of("connector", "kafka"))).isNull(); + } + + @Test + void returnsNullWhenConnectorOptionMissing() { + assertThat(DatasetIdentityRegistry.resolve("t", Map.of())).isNull(); + } + + @Test + void returnsNullWhenRequiredOptionMissing() { + assertThat(DatasetIdentityRegistry.resolve("t", Map.of("connector", "mysql-cdc", "hostname", "h"))) + .isNull(); + } +} diff --git a/streampark-flink/streampark-flink-shims/streampark-flink-shims-base/src/test/java/org/apache/streampark/flink/core/lineage/SqlWithOptionsParserTest.java b/streampark-flink/streampark-flink-shims/streampark-flink-shims-base/src/test/java/org/apache/streampark/flink/core/lineage/SqlWithOptionsParserTest.java new file mode 100644 index 0000000000..6f83c5b941 --- /dev/null +++ b/streampark-flink/streampark-flink-shims/streampark-flink-shims-base/src/test/java/org/apache/streampark/flink/core/lineage/SqlWithOptionsParserTest.java @@ -0,0 +1,159 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one or more + * contributor license agreements. See the NOTICE file distributed with + * this work for additional information regarding copyright ownership. + * The ASF licenses this file to You under the Apache License, Version 2.0 + * (the "License"); you may not use this file except in compliance with + * the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +package org.apache.streampark.flink.core.lineage; + +import org.junit.jupiter.api.Test; + +import static org.assertj.core.api.Assertions.assertThat; + +class SqlWithOptionsParserTest { + + @Test + void parsesPlainCreateTableWithOptions() { + // Deliberately NOT "CREATE TEMPORARY TABLE" — StreamPark SQL jobs commonly omit TEMPORARY, + // and this parser must not silently lose lineage for that common case. + String sql = + "CREATE TABLE mysql_pat_surgery (id BIGINT, name STRING) WITH (" + + "'connector' = 'mysql-cdc'," + + "'hostname' = '192.168.10.131'," + + "'port' = '3306'," + + "'database-name' = 'lineage_flink_verify'," + + "'table-name' = 'pat_surgery')"; + + SqlWithOptionsParser.WithOptions result = SqlWithOptionsParser.parse(sql); + + assertThat(result).isNotNull(); + assertThat(result.name()).isEqualTo("mysql_pat_surgery"); + assertThat(result.options()) + .containsEntry("connector", "mysql-cdc") + .containsEntry("hostname", "192.168.10.131") + .containsEntry("port", "3306") + .containsEntry("database-name", "lineage_flink_verify") + .containsEntry("table-name", "pat_surgery"); + } + + @Test + void parsesCreateTemporaryTableWithOptions() { + String sql = + "CREATE TEMPORARY TABLE IF NOT EXISTS `doris_sink` (id BIGINT) WITH (" + + "'connector' = 'doris'," + + "'fenodes' = '192.168.10.131:8030'," + + "'table.identifier' = 'db.ods_table')"; + + SqlWithOptionsParser.WithOptions result = SqlWithOptionsParser.parse(sql); + + assertThat(result).isNotNull(); + assertThat(result.name()).isEqualTo("doris_sink"); + assertThat(result.options()).containsEntry("connector", "doris"); + } + + @Test + void parsesLowercaseAndQualifiedTableNames() { + SqlWithOptionsParser.WithOptions result = + SqlWithOptionsParser.parse("create table mydb.mytable (id BIGINT) with ('connector' = 'doris')"); + + assertThat(result).isNotNull(); + assertThat(result.name()).isEqualTo("mytable"); + assertThat(result.options()).containsEntry("connector", "doris"); + } + + @Test + void parsesLongOptionValueWithoutOverflowingTheStack() { + // A greedily-repeated literal body recurses once per character, so a long value (a + // certificate, a serialized properties blob) used to throw StackOverflowError on the + // submission path rather than yielding lineage. + StringBuilder longValue = new StringBuilder(); + for (int i = 0; i < 20000; i++) { + longValue.append('a'); + } + String sql = + "CREATE TABLE t (id BIGINT) WITH ('connector' = 'mysql-cdc','password' = '" + longValue + "')"; + + SqlWithOptionsParser.WithOptions result = SqlWithOptionsParser.parse(sql); + + assertThat(result).isNotNull(); + assertThat(result.options().get("password")).hasSize(20000); + } + + @Test + void unescapesDoubledSingleQuotesInsideOptionValues() { + String sql = + "CREATE TABLE t (id BIGINT) WITH (" + + "'connector' = 'mysql-cdc'," + + "'password' = 'a''b')"; + + SqlWithOptionsParser.WithOptions result = SqlWithOptionsParser.parse(sql); + + assertThat(result).isNotNull(); + assertThat(result.options()).containsEntry("password", "a'b"); + } + + @Test + void reducesQualifiedTableNameToItsLocalName() { + // A CompiledPlan identifier always reports the table under its own name only, so keying + // these options by the qualified form would never match and the table would be treated as + // catalog-backed instead of connector-backed. + String sql = + "CREATE TABLE mydb.mytable (id BIGINT) WITH ('connector' = 'doris'," + + "'fenodes' = '192.168.10.131:8030','table.identifier' = 'db.t')"; + + SqlWithOptionsParser.WithOptions result = SqlWithOptionsParser.parse(sql); + + assertThat(result).isNotNull(); + assertThat(result.name()).isEqualTo("mytable"); + assertThat(result.options()).containsEntry("connector", "doris"); + } + + @Test + void ignoresAWithClauseThatOnlyAppearsInsideAStringLiteral() { + String sql = + "CREATE TABLE t (id BIGINT COMMENT 'joined WITH (other)') WITH (" + + "'connector' = 'mysql-cdc','hostname' = 'h')"; + + SqlWithOptionsParser.WithOptions result = SqlWithOptionsParser.parse(sql); + + assertThat(result).isNotNull(); + assertThat(result.options()) + .containsEntry("connector", "mysql-cdc") + .containsEntry("hostname", "h"); + } + + @Test + void parsesCreateCatalogType() { + String sql = + "CREATE CATALOG paimon_s3 WITH ('type' = 'paimon','warehouse' = 's3://bucket/warehouse')"; + + SqlWithOptionsParser.WithOptions result = SqlWithOptionsParser.parseCatalog(sql); + + assertThat(result).isNotNull(); + assertThat(result.name()).isEqualTo("paimon_s3"); + assertThat(result.options()).containsEntry("type", "paimon"); + } + + @Test + void returnsNullWhenStatementHasNoWithClause() { + String sql = "CREATE TABLE t LIKE other_table"; + + assertThat(SqlWithOptionsParser.parse(sql)).isNull(); + } + + @Test + void returnsNullForNonCreateTableStatement() { + assertThat(SqlWithOptionsParser.parse("INSERT INTO sink SELECT * FROM src")).isNull(); + } +}